feat(s31): --semantic-retrieval opt-in threaded through run_project/run_portfolio

This commit is contained in:
Kjell Tore Guttormsen 2026-07-25 06:29:39 +02:00
commit 63734f5bfa
2 changed files with 195 additions and 1 deletions

View file

@ -50,6 +50,11 @@ from portfolio_optimiser.provenance import ProvenanceStamp
from portfolio_optimiser.reference_domain import Project, load_reference_projects from portfolio_optimiser.reference_domain import Project, load_reference_projects
from portfolio_optimiser.validator import Rejection, ValidatedProposal from portfolio_optimiser.validator import Rejection, ValidatedProposal
from portfolio_optimiser import okf, outbox from portfolio_optimiser import okf, outbox
from portfolio_optimiser.semretrieval import (
SEMANTIC_WEIGHT_DEFAULT,
FakeEmbedder,
HybridRanker,
)
from portfolio_optimiser.verdicts import ( from portfolio_optimiser.verdicts import (
ExpeLContextProvider, ExpeLContextProvider,
ProposalFeatures, ProposalFeatures,
@ -58,6 +63,7 @@ from portfolio_optimiser.verdicts import (
bundle_candidate_features, bundle_candidate_features,
capture_verdict, capture_verdict,
load_verdicts_from_dir, load_verdicts_from_dir,
similarity,
) )
from portfolio_optimiser.value_report import ( from portfolio_optimiser.value_report import (
build_value_report, build_value_report,
@ -251,6 +257,7 @@ async def run_project(
notify: Callable[[Verdict], None] | None = None, notify: Callable[[Verdict], None] | None = None,
meter: TokenMeter | None = None, meter: TokenMeter | None = None,
live_dry_run: bool = False, live_dry_run: bool = False,
semantic_retrieval: bool = False,
) -> RunResult | DryRunReport: ) -> RunResult | DryRunReport:
"""Run the vertical slice for ONE project. ``client_factory`` is the test-injection seam """Run the vertical slice for ONE project. ``client_factory`` is the test-injection seam
(defaults to the real backend). ``verdict_input`` carries the expert decision/rationale (defaults to the real backend). ``verdict_input`` carries the expert decision/rationale
@ -271,7 +278,12 @@ async def run_project(
crossed, and ``ValueError`` when ``outbox_dir`` is set without a ``run_id``. ``live_dry_run`` crossed, and ``ValueError`` when ``outbox_dir`` is set without a ``run_id``. ``live_dry_run``
(S4.2, comparison protocol §4 pkt 2/3) is the offline drill: it walks the whole path up to the (S4.2, comparison protocol §4 pkt 2/3) is the offline drill: it walks the whole path up to the
EAGER client build, writes the run-config artefact (when ``outbox_dir`` is set), and returns a EAGER client build, writes the run-config artefact (when ``outbox_dir`` is set), and returns a
``DryRunReport`` BEFORE the first model call (``debate.run``) zero chat calls.""" ``DryRunReport`` BEFORE the first model call (``debate.run``) zero chat calls.
``semantic_retrieval`` (S3.1) is the opt-in scaling seam: when true, the store's ranker is
swapped for a ``HybridRanker`` that blends brute-force cosine over embedded features with the
structural score, so a semantically related prior verdict carrying a DIFFERENT cost-code set
can reach the Step-1 fold. Default false keeps the structural, text-excluded ranking exactly
as before."""
# 0. Fail-fast: an outbox write is byte-deterministic and keyed on run_id — no wall-clock default. # 0. Fail-fast: an outbox write is byte-deterministic and keyed on run_id — no wall-clock default.
if outbox_dir is not None and run_id is None: if outbox_dir is not None and run_id is None:
raise ValueError( raise ValueError(
@ -376,6 +388,13 @@ async def run_project(
# post-generation into a discarded SessionContext (step 7 below), so a prior verdict could not # post-generation into a discarded SessionContext (step 7 below), so a prior verdict could not
# reach the next hypothesis. Bundle-driven path with a populated store only; the road path is # reach the next hypothesis. Bundle-driven path with a populated store only; the road path is
# untouched (its post-hoc, proposal-keyed retrieval below is unchanged). # untouched (its post-hoc, proposal-keyed retrieval below is unchanged).
# S3.1 opt-in: install the hybrid ranker on the RESOLVED store, immediately before the fold it
# is meant to affect. Anchoring matters — set at the verdict_dir-conditional resolution above
# it would miss every run without an inbox, and set after the fold it would rank nothing that
# reaches this prompt. Flag off => store.retriever stays None => StructuralRetriever default.
if semantic_retrieval and store is not None:
store.retriever = HybridRanker(FakeEmbedder(), similarity, SEMANTIC_WEIGHT_DEFAULT)
if bundle_dir is not None and store is not None and store.verdicts: if bundle_dir is not None and store is not None and store.verdicts:
expel_query = bundle_candidate_features(bundle_dir) expel_query = bundle_candidate_features(bundle_dir)
fewshot = ExpeLContextProvider(store, expel_query, k=top_k).format_fewshot() fewshot = ExpeLContextProvider(store, expel_query, k=top_k).format_fewshot()
@ -526,6 +545,7 @@ async def run_portfolio(
max_tokens: int = 100_000, max_tokens: int = 100_000,
top_k: int = 3, top_k: int = 3,
meter_factory: Callable[[], TokenMeter] | None = None, meter_factory: Callable[[], TokenMeter] | None = None,
semantic_retrieval: bool = False,
) -> PortfolioResult: ) -> PortfolioResult:
"""Fan out over a portfolio of independent projects SEQUENTIALLY, composing ``run_project`` """Fan out over a portfolio of independent projects SEQUENTIALLY, composing ``run_project``
as-is (every project's execution state — meter, debate, retrieval context — is built fresh as-is (every project's execution state — meter, debate, retrieval context — is built fresh
@ -551,6 +571,10 @@ async def run_portfolio(
projects = {p.id: p for p in load_reference_projects()} projects = {p.id: p for p in load_reference_projects()}
ids = list(project_ids) if project_ids is not None else list(projects) ids = list(project_ids) if project_ids is not None else list(projects)
store = store if store is not None else VerdictStore(verdicts=[]) store = store if store is not None else VerdictStore(verdicts=[])
# S3.1: one shared store is threaded across every project, so installing the hybrid here —
# before the loop — covers every project's Step-1 fold in this pass.
if semantic_retrieval:
store.retriever = HybridRanker(FakeEmbedder(), similarity, SEMANTIC_WEIGHT_DEFAULT)
ledger = ledger if ledger is not None else SavingsLedger(entries=[]) ledger = ledger if ledger is not None else SavingsLedger(entries=[])
goals = goals if goals is not None else GoalConfig() goals = goals if goals is not None else GoalConfig()
portfolio_baseline_ore = _to_ore(sum(projects[p].total_cost for p in ids if p in projects)) portfolio_baseline_ore = _to_ore(sum(projects[p].total_cost for p in ids if p in projects))
@ -673,6 +697,14 @@ def main(argv: list[str] | None = None) -> int:
help="portfolio mode: accumulated savings ledger JSON (fail-fast) read for the goal-stop " help="portfolio mode: accumulated savings ledger JSON (fail-fast) read for the goal-stop "
"(earlier out-of-band HITL realizations — never built during the pass)", "(earlier out-of-band HITL realizations — never built during the pass)",
) )
parser.add_argument(
"--semantic-retrieval",
action="store_true",
help="S3.1 opt-in scaling seam: rank prior verdicts with a hybrid of brute-force cosine "
"over embedded features and the structural score, so a semantically related verdict with a "
"DIFFERENT cost-code set can reach the hypothesis prompt. Valid in both modes; OFF by "
"default, and off means the structural, text-excluded ranking is unchanged",
)
parser.add_argument("--decision", default="approved", choices=["approved", "rejected"]) parser.add_argument("--decision", default="approved", choices=["approved", "rejected"])
parser.add_argument("--rationale", default="reviewed by expert") parser.add_argument("--rationale", default="reviewed by expert")
parser.add_argument( parser.add_argument(
@ -721,6 +753,7 @@ def main(argv: list[str] | None = None) -> int:
"--outbox-dir": args.outbox_dir is not None, "--outbox-dir": args.outbox_dir is not None,
"--run-id": args.run_id is not None, "--run-id": args.run_id is not None,
"--dimension-config": args.dimension_config is not None, "--dimension-config": args.dimension_config is not None,
"--semantic-retrieval": args.semantic_retrieval,
} }
if any(report_forbidden.values()): if any(report_forbidden.values()):
print( print(
@ -792,6 +825,7 @@ def main(argv: list[str] | None = None) -> int:
dimension=dimension, dimension=dimension,
ledger=ledger, ledger=ledger,
goals=goals, goals=goals,
semantic_retrieval=args.semantic_retrieval,
) )
) )
except (ValueError, FileNotFoundError, ValidationError) as exc: except (ValueError, FileNotFoundError, ValidationError) as exc:
@ -878,6 +912,7 @@ def main(argv: list[str] | None = None) -> int:
outbox_dir=args.outbox_dir, outbox_dir=args.outbox_dir,
run_id=args.run_id, run_id=args.run_id,
verdict_input={"decision": args.decision, "rationale": args.rationale}, verdict_input={"decision": args.decision, "rationale": args.rationale},
semantic_retrieval=args.semantic_retrieval,
) )
), ),
) )

View file

@ -17,6 +17,7 @@ import pytest
from portfolio_optimiser import run from portfolio_optimiser import run
from portfolio_optimiser.dimension import Dimension from portfolio_optimiser.dimension import Dimension
from portfolio_optimiser.ledger import LedgerEntry, SavingsLedger from portfolio_optimiser.ledger import LedgerEntry, SavingsLedger
from portfolio_optimiser.run import run_project
from portfolio_optimiser.verdicts import ProposalFeatures, capture_verdict, write_verdict from portfolio_optimiser.verdicts import ProposalFeatures, capture_verdict, write_verdict
BUNDLE_DIR = Path(__file__).resolve().parents[1] / "shared" / "examples" / "bygg-energi-mikro" BUNDLE_DIR = Path(__file__).resolve().parents[1] / "shared" / "examples" / "bygg-energi-mikro"
@ -416,3 +417,161 @@ def test_json_without_report_refuses(tmp_path, capsys) -> None:
rc = run.main(["--json", "--ledger", str(_report_ledger(tmp_path))]) rc = run.main(["--json", "--ledger", str(_report_ledger(tmp_path))])
assert rc == 1 assert rc == 1
assert "refused" in capsys.readouterr().err.lower() assert "refused" in capsys.readouterr().err.lower()
# --- S3.1 SC6: --semantic-retrieval opt-in (default OFF) --------------------------------------
_ENERGY_REPLY = (
'{"measure":"LED-retrofit av kontorbelysning","affected_items":'
'[{"code":"ENERGI-TOTAL-EL","quantity":300000,"unit_cost":1.0}],"claimed_saving_nok":30000}'
)
_VERDICT_INPUT = {"decision": "approved", "rationale": "expert reviewed (sim)"}
# The marker lives in the RATIONALE, which is what ``format_fewshot`` emits into the prompt.
# "0.37" appears nowhere in the bundle (grep-verified), so its presence in a prompt can only have
# come through the ExpeL fold — it cannot be leaked by bundle context.
_TIE_MARKER = "realiseringsgrad=0.37"
_MARKER_ID = "zz-s31-run-marker"
_DISTRACTOR_ID = "aa-s31-run-distractor"
def _tied_pair_store():
"""Two verdicts that are structurally IDENTICAL to the bundle's candidate (similarity 1.0
each), so the structural ranking can only separate them on ``id`` and ``_MARKER_ID`` loses
that tie. Only the cosine term can promote it.
The bundle's own seed verdict is deliberately NOT included: its features are byte-identical to
the query, making it a perfect cosine match that would win every hybrid ranking and so could
never demonstrate a tie-break."""
from portfolio_optimiser.verdicts import Verdict, VerdictStore, bundle_candidate_features
query = bundle_candidate_features(str(BUNDLE_DIR))
def tied(verdict_id: str, description: str, rationale: str) -> Verdict:
return Verdict(
id=verdict_id,
proposal_features=ProposalFeatures(
affected_codes=query.affected_codes,
measure_type=query.measure_type,
claimed_saving_nok=query.claimed_saving_nok,
description=description,
),
decision="approved",
rationale=rationale,
)
return VerdictStore(
verdicts=[
tied(
_DISTRACTOR_ID,
"avvist forslag om reforhandling av renholdskontrakt i administrasjonsbygget",
"ingen realiseringsdata for dette tiltaket",
),
tied(
_MARKER_ID,
"LED-retrofit i kontorlokaler: 90 W armaturer erstattet med 40 W",
f"tidligere LED-dom [{_TIE_MARKER}]",
),
]
)
def _generation_prompts(sink: list[str]) -> list[str]:
return [p for p in sink if "SavingsProposal" in p]
def test_semantic_retrieval_flag_parses_offline(capsys) -> None:
"""(a) the flag is accepted in single-project mode and the run still stops offline."""
rc = run.main(
[
_PID,
"--docs-dir",
str(BUNDLE_DIR),
"--bundle-dir",
str(BUNDLE_DIR),
"--semantic-retrieval",
"--live-dry-run",
]
)
assert rc == 0
assert "LIVE-DRY-RUN OK" in capsys.readouterr().out
def test_semantic_retrieval_is_not_refused_in_portfolio_mode(capsys) -> None:
"""The flag is valid in BOTH modes (like --dimension-config), so the portfolio partition must
not name it. Probed via a run that IS refused for a different flag: the refusal lists
``--docs-dir`` and must NOT mention ``--semantic-retrieval``."""
rc = run.main(["--portfolio", "--semantic-retrieval", "--docs-dir", str(BUNDLE_DIR)])
err = capsys.readouterr().err
assert rc == 1
assert "--docs-dir" in err
assert "--semantic-retrieval" not in err
def test_report_with_semantic_retrieval_is_refused(tmp_path, capsys) -> None:
"""--report is an ALLOWLIST: only --ledger/--json ride along. A silently-dropped
--semantic-retrieval would break that partition."""
ledger_file = tmp_path / "ledger.json"
SavingsLedger(entries=[]).save(str(ledger_file))
rc = run.main(["--report", "--ledger", str(ledger_file), "--semantic-retrieval"])
assert rc == 1
assert "refused" in capsys.readouterr().err.lower()
async def test_semantic_retrieval_on_swaps_the_fewshot_reaching_the_prompt(
make_recording_client_factory,
) -> None:
"""(c) the substantive run-level proof: with ``semantic_retrieval=True`` and ``top_k=1``, the
verdict that only cosine can surface is the one whose rationale reaches the hypothesis prompt.
Drives the REAL Step-1 fold via the recording client not ``--live-dry-run``, which returns
before the fold."""
factory, recorded = make_recording_client_factory(_ENERGY_REPLY)
await run_project(
_PID,
"local",
docs_dir=str(BUNDLE_DIR),
bundle_dir=str(BUNDLE_DIR),
verdict_input=_VERDICT_INPUT,
store=_tied_pair_store(),
client_factory=factory,
top_k=1,
semantic_retrieval=True,
)
gen_prompts = _generation_prompts(recorded)
assert gen_prompts, "the generation call must have happened"
assert any(_TIE_MARKER in p for p in gen_prompts), (
"the cosine-surfaced verdict did not reach the hypothesis prompt — "
"--semantic-retrieval is not installing the HybridRanker before the Step-1 fold"
)
assert any(_MARKER_ID in p for p in gen_prompts)
async def test_semantic_retrieval_off_leaves_the_structural_pick_in_the_prompt(
make_recording_client_factory,
) -> None:
"""CAUSALITY CONTROL — the identical run with the flag OFF must carry the STRUCTURAL winner
instead, and no marker. This is what makes the positive above load-bearing: the swap is caused
by the flag, not by the fixture merely containing the marker."""
factory, recorded = make_recording_client_factory(_ENERGY_REPLY)
await run_project(
_PID,
"local",
docs_dir=str(BUNDLE_DIR),
bundle_dir=str(BUNDLE_DIR),
verdict_input=_VERDICT_INPUT,
store=_tied_pair_store(),
client_factory=factory,
top_k=1,
)
assert all(_TIE_MARKER not in p for p in recorded), (
"the marker reached a prompt with --semantic-retrieval OFF — the default path is not the "
"structural ranking, or the assertion is not load-bearing"
)
gen_prompts = _generation_prompts(recorded)
assert any(_DISTRACTOR_ID in p for p in gen_prompts), (
"the structural winner did not reach the prompt — the default fold is broken"
)