test(s31): close 1 review MAJOR — main()-level load-bearing proof of the --semantic-retrieval wiring
This commit is contained in:
parent
8e9f6603d7
commit
cabe05acf3
1 changed files with 228 additions and 9 deletions
|
|
@ -1,15 +1,35 @@
|
||||||
"""Load-bearing CLI test for the observable GoalReached stop (S5.3 Step 3, SC1).
|
"""Load-bearing CLI tests: the observable GoalReached stop (S5.3 Step 3, SC1) and the
|
||||||
|
``--semantic-retrieval`` wiring proof (S3.1 remediation, review finding ``c23e87ee``).
|
||||||
|
|
||||||
DETACH THAT TURNS THIS RED: route ``--goals``/``--ledger`` away from ``run_portfolio`` (drop the
|
DETACH THAT TURNS THE GOAL ARMS RED: route ``--goals``/``--ledger`` away from ``run_portfolio``
|
||||||
wiring in ``main()``'s portfolio dispatch) → no goal-stop line reaches stdout → the positive arm's
|
(drop the wiring in ``main()``'s portfolio dispatch) → no goal-stop line reaches stdout → the
|
||||||
marker assertion fails. The positive+control pair proves the printed scope/øre fields flow from
|
positive arm's marker assertion fails. The positive+control pair proves the printed scope/øre
|
||||||
``run_portfolio``'s actual ``GoalReached``/``PortfolioResult`` values — a canned print could not
|
fields flow from ``run_portfolio``'s actual ``GoalReached``/``PortfolioResult`` values — a canned
|
||||||
produce BOTH the portfolio-scope and the project-scope variant from the same wiring.
|
print could not produce BOTH the portfolio-scope and the project-scope variant from the same
|
||||||
|
wiring.
|
||||||
|
|
||||||
Both arms are provably OFFLINE: the goal is already met before any project builds a chat client
|
DETACH POINTS THAT TURN THE ``--semantic-retrieval`` ARMS RED — two, because the review's finding
|
||||||
|
was that the flag was green-but-dead at exactly these joints:
|
||||||
|
1. Hardcode ``semantic_retrieval=False`` at ``main()``'s full-run ``run_project(...)`` call → the
|
||||||
|
single-project positive loses its marker → RED.
|
||||||
|
2. Drop ``semantic_retrieval=semantic_retrieval`` from the ``run_project(...)`` call inside
|
||||||
|
``run_portfolio`` → the portfolio positive loses its marker → RED.
|
||||||
|
Both were previously covered ONLY below the CLI (``run_project`` driven directly), so either
|
||||||
|
mutation left the suite green. These arms drive ``run.main([...])`` and nothing else.
|
||||||
|
|
||||||
|
Why the fixture looks the way it does: ``main()`` exposes no ``--top-k``, so retrieval returns 3
|
||||||
|
verdicts. The inbox therefore holds FOUR structurally tied verdicts — three decoys whose minted ids
|
||||||
|
sort before the marker's, plus the marker. Structurally (score tied, ``id`` ascending) the marker is
|
||||||
|
the FOURTH and never reaches the prompt; only the cosine term lifts it into the top 3. A two-verdict
|
||||||
|
fixture would put the marker in the prompt either way and the control would prove nothing.
|
||||||
|
|
||||||
|
Both goal arms are provably OFFLINE: the goal is already met before any project builds a chat client
|
||||||
(portfolio-hard ``break``s at ``run.py:560-567``; per-project-hard ``continue``s past the only pid at
|
(portfolio-hard ``break``s at ``run.py:560-567``; per-project-hard ``continue``s past the only pid at
|
||||||
``run.py:571-579`` — no client is ever constructed). Marker value 13731 øre appears nowhere else in
|
``run.py:571-579`` — no client is ever constructed). The ``--semantic-retrieval`` arms DO run a full
|
||||||
the codebase (repo marker convention). No socket/network is exercised (brief NFR).
|
pass, but every model call goes to a recording stand-in injected by monkeypatching
|
||||||
|
``run._default_factory`` — the seam ``main()`` resolves through, since it never passes
|
||||||
|
``client_factory``. No socket/network is exercised (brief NFR). Marker value 13731 øre and the
|
||||||
|
``SENTINEL-CLI-45f9a2`` string each appear nowhere else in the codebase (repo marker convention).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
@ -21,9 +41,31 @@ import pytest
|
||||||
|
|
||||||
from portfolio_optimiser import run
|
from portfolio_optimiser import run
|
||||||
from portfolio_optimiser.ledger import LedgerEntry, SavingsLedger
|
from portfolio_optimiser.ledger import LedgerEntry, SavingsLedger
|
||||||
|
from portfolio_optimiser.reference_domain import Project
|
||||||
|
from portfolio_optimiser.verdicts import ProposalFeatures, capture_verdict, write_verdict
|
||||||
|
|
||||||
_MARKER_ORE = 13731 # marker: appears nowhere else in the codebase
|
_MARKER_ORE = 13731 # marker: appears nowhere else in the codebase
|
||||||
|
|
||||||
|
_BUNDLE_DIR = Path(__file__).resolve().parents[1] / "shared" / "examples" / "bygg-energi-mikro"
|
||||||
|
_PID = "BYGG-KONTOR-NORD"
|
||||||
|
|
||||||
|
# The learning signal rides in the RATIONALE — that is what ``ExpeLContextProvider.format_fewshot``
|
||||||
|
# emits into the prompt. It cannot be leaked by bundle context: the string appears nowhere in the
|
||||||
|
# bundle (grep-verified), so its presence in a generation prompt can only have come via the fold.
|
||||||
|
_CLI_MARKER = "SENTINEL-CLI-45f9a2 realiseringsgrad=0.5391"
|
||||||
|
|
||||||
|
# Extra cost codes appended to the bundle's own code. Every combination ties on the structural
|
||||||
|
# score (equal Jaccard, same measure, same magnitude bucket); they differ only in the code set, so
|
||||||
|
# the canonical embedding string differs and cosine has something to separate. Chosen empirically:
|
||||||
|
# the three decoys' minted ids all sort BEFORE the marker's, and the marker's cosine is the highest.
|
||||||
|
_MARKER_CODE = "45.9"
|
||||||
|
_DECOY_CODES = ("58.9", "15.5", "36.8")
|
||||||
|
|
||||||
|
_ENERGY_REPLY = (
|
||||||
|
'{"measure":"LED-retrofit av kontorbelysning","affected_items":'
|
||||||
|
'[{"code":"ENERGI-TOTAL-EL","quantity":300000,"unit_cost":1.0}],"claimed_saving_nok":30000}'
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture(autouse=True)
|
@pytest.fixture(autouse=True)
|
||||||
def _isolate_model_env(monkeypatch: pytest.MonkeyPatch) -> None:
|
def _isolate_model_env(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||||
|
|
@ -85,3 +127,180 @@ def test_per_project_hard_goal_control_distinguishes_scope(tmp_path, capsys) ->
|
||||||
assert "scope=project" in out
|
assert "scope=project" in out
|
||||||
assert "stopped_early=False" in out
|
assert "stopped_early=False" in out
|
||||||
assert "scope=portfolio" not in out
|
assert "scope=portfolio" not in out
|
||||||
|
|
||||||
|
|
||||||
|
# --- S3.1: the --semantic-retrieval wiring, proved at main() level (finding c23e87ee) -----------
|
||||||
|
|
||||||
|
|
||||||
|
def _generation_prompts(sink: list[str]) -> list[str]:
|
||||||
|
"""The generation-call prompts (``generate._build_messages`` embeds 'SavingsProposal'),
|
||||||
|
isolated from the debate-round prompts also captured in the shared sink."""
|
||||||
|
return [p for p in sink if "SavingsProposal" in p]
|
||||||
|
|
||||||
|
|
||||||
|
def _tied_inbox(tmp_path: Path) -> str:
|
||||||
|
"""A populated async inbox: three structurally tied decoys plus the marker verdict.
|
||||||
|
|
||||||
|
Written through the real authoring primitives — ``write_verdict(capture_verdict(...))`` — so
|
||||||
|
every id is minted the way the framework mints them, never hand-assigned. ``--verdict-dir`` is
|
||||||
|
the ONLY route by which ``main()`` can hand ``run_project`` a non-empty store."""
|
||||||
|
from portfolio_optimiser.verdicts import bundle_candidate_features
|
||||||
|
|
||||||
|
query = bundle_candidate_features(str(_BUNDLE_DIR))
|
||||||
|
inbox = tmp_path / "inbox"
|
||||||
|
|
||||||
|
def tied(extra_code: str, decision: str, rationale: str) -> None:
|
||||||
|
write_verdict(
|
||||||
|
str(inbox),
|
||||||
|
capture_verdict(
|
||||||
|
ProposalFeatures(
|
||||||
|
affected_codes=query.affected_codes | {extra_code},
|
||||||
|
measure_type=query.measure_type,
|
||||||
|
claimed_saving_nok=query.claimed_saving_nok,
|
||||||
|
# ``description == measure`` is the shape both live minting paths emit.
|
||||||
|
description=query.measure_type,
|
||||||
|
),
|
||||||
|
decision,
|
||||||
|
rationale,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
for code in _DECOY_CODES:
|
||||||
|
tied(code, "rejected", f"ingen realiseringsdata for tiltak {code}")
|
||||||
|
tied(_MARKER_CODE, "approved", f"tidligere LED-dom [{_CLI_MARKER}]")
|
||||||
|
return str(inbox)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture()
|
||||||
|
def _recorded_main(monkeypatch):
|
||||||
|
"""Inject a recording chat client into the seam ``main()`` actually resolves through.
|
||||||
|
|
||||||
|
``main()`` never passes ``client_factory``, so ``run_project`` falls back to the module-level
|
||||||
|
``_default_factory``. Patching it is therefore the only observation point available to a
|
||||||
|
main()-level test — and patching it (rather than passing a factory) is what keeps this test
|
||||||
|
honest about the CLI path."""
|
||||||
|
from tests.conftest import _RecordingChatClient
|
||||||
|
|
||||||
|
sink: list[str] = []
|
||||||
|
monkeypatch.setattr(
|
||||||
|
"portfolio_optimiser.run._default_factory",
|
||||||
|
lambda profile: lambda role: _RecordingChatClient(sink, _ENERGY_REPLY),
|
||||||
|
)
|
||||||
|
return sink
|
||||||
|
|
||||||
|
|
||||||
|
def test_cli_semantic_retrieval_reaches_the_hypothesis_prompt(tmp_path, _recorded_main) -> None:
|
||||||
|
"""POSITIVE — ``run.main([... --semantic-retrieval])`` must carry the cosine-surfaced verdict's
|
||||||
|
rationale into the hypothesis prompt.
|
||||||
|
|
||||||
|
This is the arm the review's ``c23e87ee`` demanded: every prior proof drove ``run_project``
|
||||||
|
directly, so the CLI could stop forwarding the flag without a single test noticing.
|
||||||
|
|
||||||
|
Detach point: hardcode ``semantic_retrieval=False`` at ``main()``'s ``run_project(...)`` call
|
||||||
|
→ RED."""
|
||||||
|
rc = run.main(
|
||||||
|
[
|
||||||
|
_PID,
|
||||||
|
"--docs-dir",
|
||||||
|
str(_BUNDLE_DIR),
|
||||||
|
"--bundle-dir",
|
||||||
|
str(_BUNDLE_DIR),
|
||||||
|
"--verdict-dir",
|
||||||
|
_tied_inbox(tmp_path),
|
||||||
|
"--semantic-retrieval",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
assert rc == 0
|
||||||
|
|
||||||
|
gen_prompts = _generation_prompts(_recorded_main)
|
||||||
|
assert gen_prompts, "the generation call must have happened"
|
||||||
|
assert any(_CLI_MARKER in p for p in gen_prompts), (
|
||||||
|
"the cosine-surfaced verdict did not reach the hypothesis prompt — main() is not "
|
||||||
|
"forwarding --semantic-retrieval into the Step-1 fold"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_cli_without_the_flag_leaves_the_marker_out(tmp_path, _recorded_main) -> None:
|
||||||
|
"""CAUSALITY CONTROL — the identical CLI invocation WITHOUT the flag must not carry the marker.
|
||||||
|
|
||||||
|
Without this the positive proves nothing: it would pass merely because the inbox contains the
|
||||||
|
marker. Structurally the marker is the fourth of four tied verdicts and retrieval returns three,
|
||||||
|
so only the cosine term can reach it."""
|
||||||
|
rc = run.main(
|
||||||
|
[
|
||||||
|
_PID,
|
||||||
|
"--docs-dir",
|
||||||
|
str(_BUNDLE_DIR),
|
||||||
|
"--bundle-dir",
|
||||||
|
str(_BUNDLE_DIR),
|
||||||
|
"--verdict-dir",
|
||||||
|
_tied_inbox(tmp_path),
|
||||||
|
]
|
||||||
|
)
|
||||||
|
assert rc == 0
|
||||||
|
|
||||||
|
assert all(_CLI_MARKER not in p for p in _recorded_main), (
|
||||||
|
"the marker reached a prompt with --semantic-retrieval OFF — the default CLI path is not "
|
||||||
|
"the structural ranking, so the positive arm is not load-bearing"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_cli_portfolio_mode_forwards_semantic_retrieval(tmp_path, monkeypatch, _recorded_main):
|
||||||
|
"""POSITIVE (portfolio arm) — ``main(["--portfolio", "--semantic-retrieval"])`` must reach each
|
||||||
|
project's Step-1 fold.
|
||||||
|
|
||||||
|
Portfolio mode refuses ``--verdict-dir`` (it is single-project-only), so the inbox arrives on
|
||||||
|
the ``Project`` record instead — the same topology
|
||||||
|
``tests/test_portfolio_learning_loadbearing.py`` uses.
|
||||||
|
|
||||||
|
Detach point: drop ``semantic_retrieval=semantic_retrieval`` from the ``run_project(...)`` call
|
||||||
|
inside ``run_portfolio`` → RED."""
|
||||||
|
project = _bundle_project(tmp_path, verdict_dir=_tied_inbox(tmp_path))
|
||||||
|
monkeypatch.setattr("portfolio_optimiser.run.load_reference_projects", lambda: (project,))
|
||||||
|
|
||||||
|
rc = run.main(["--portfolio", "--semantic-retrieval"])
|
||||||
|
assert rc == 0
|
||||||
|
|
||||||
|
gen_prompts = _generation_prompts(_recorded_main)
|
||||||
|
assert gen_prompts, "the generation call must have happened"
|
||||||
|
assert any(_CLI_MARKER in p for p in gen_prompts), (
|
||||||
|
"the marker did not reach the hypothesis prompt in portfolio mode — run_portfolio is not "
|
||||||
|
"forwarding semantic_retrieval to run_project"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_cli_portfolio_mode_without_the_flag_leaves_the_marker_out(
|
||||||
|
tmp_path, monkeypatch, _recorded_main
|
||||||
|
):
|
||||||
|
"""CAUSALITY CONTROL for the portfolio arm."""
|
||||||
|
project = _bundle_project(tmp_path, verdict_dir=_tied_inbox(tmp_path))
|
||||||
|
monkeypatch.setattr("portfolio_optimiser.run.load_reference_projects", lambda: (project,))
|
||||||
|
|
||||||
|
rc = run.main(["--portfolio"])
|
||||||
|
assert rc == 0
|
||||||
|
|
||||||
|
assert all(_CLI_MARKER not in p for p in _recorded_main), (
|
||||||
|
"the marker reached a prompt without the flag — the portfolio default is not structural"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _bundle_project(tmp_path: Path, *, verdict_dir: str) -> Project:
|
||||||
|
"""One bundle-backed project pointed at the SAME bundle the tied set was derived from — a
|
||||||
|
different bundle would yield a different query and untie the set."""
|
||||||
|
docs = tmp_path / "portfolio-docs"
|
||||||
|
docs.mkdir(exist_ok=True)
|
||||||
|
(docs / "cost.txt").write_text(
|
||||||
|
"LED-retrofit av lysrorarmaturer i kontorlokaler reduserte energikostnaden.",
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
return Project(
|
||||||
|
id=_PID,
|
||||||
|
name="Bundle-backed portfolio project",
|
||||||
|
description="bundle-backed project for the CLI semantic-retrieval forwarding proof",
|
||||||
|
currency="NOK",
|
||||||
|
cost_items=(),
|
||||||
|
docs_dir=str(docs),
|
||||||
|
verdict_input={"decision": "approved", "rationale": "expert reviewed (sim)"},
|
||||||
|
bundle_dir=str(_BUNDLE_DIR),
|
||||||
|
verdict_dir=verdict_dir,
|
||||||
|
)
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue