feat(prepass): the debate is handed the declared cut and the navigator tools are withdrawn [skip-docs]
[skip-docs]: CLI-flagget og README-blokka kommer i steg 6/7; `prepass_payload` er foreloepig bare naabar for en bibliotek-kaller. `run_project(prepass_payload=...)` forgrener bundle-armen. UTEN payload er hver linje uendret -- pekeren, de fire verktoeyene, siteringer over hele den navigerte basen. MED et payload faar debatten et DEKLARERT KUTT og verktoeyene trekkes (SS 2.2: kontekst pre-passet holdt tilbake ble holdt tilbake med vilje; en debatt som holder BEGGE er fri til aa gaa rundt kuttet den nettopp erklaerte). Nekten PROPAGERER, aldri en stille degradering tilbake til pekeren -- `load_mandate`s regel. Maalt paa NULL modellkall, ikke paa exit-koden. `delivered == 0` nektes ved NAVN foer debatten, med nevnerne, spoersmaalet og ref-en sitert: maalt er den tilstanden bare naabar naar hvert konsept feilet leksikalsk (den andre tomme saken nekter produsenten selv), altsaa bevis for FRAVAER. Uten den falt kjoeringen gjennom til `run.py`s siteringsvakt, hvis melding navngir `docs_dir` -- som er `None` paa denne stien. `bundle_excerpt_citations` (datasource) siterer de LEVERTE konseptene alene: et stempel som siterer hele korpuset for et forslag som saa fire, gjenoppfinner den uerklaerte paastanden sømmen finnes for. Kroppen tas fra den NAVIGERTE `BundleFile`, ikke fra payloadets `text` -- den leverte teksten er NFC-normalisert med strippet hale, saa en locator over den ville ikke indeksert fila den navngir. Deler dermed ogsaa `bundle_citations`' verdict-eksklusjon i stedet for aa gjenta den. MCP-appenden ligger BEVISST under forgreningen: dette trekker navigatoerverktoeyene, ikke verktoeylista. Maalt: en tom liste naar traaden som `tools: None`, saa ingen uproevd tom-array-form innfoeres. `RunResult.prepass` og `DryRunReport.prepass` DEFAULTER (`skipped_links`-halvdelen: `None` er det sanne utsagnet "ingen payload ble gitt"), bundet i BEGGE grener saa ingen `NameError` venter paa veg-stien. `ProvenanceStamp` er BEVISST urørt -- stempelet beskriver gaten som doemte EN kandidat, dette er et RUN-nivaa-faktum om hva kjoeringen i det hele tatt fikk lese. Tilbaketrekkingen asserteres paa `fresh_workflow(tools=...)`, ALDRI paa `debate_tool_calls`: maalt er det sporet allerede tomt MED alle fire verktoeyene, fordi en `ScriptedChatClient` aldri emitterer et verktoeykall. En arm skrevet paa det kan ikke skille de to implementasjonene. GATE, IKKE VEGG: en egen arm beviser at den gatede ExpeL-folden fortsatt naar hypotese-prompten (0.82) under et payload. 1440 passed / 5 skipped (fra 1425/5, +15, 0 fjernet). ruff + mypy rene. Golden `shasum -a 1` av INNHOLDET = ea8c534773acdbe41ae68f2c55724d69aaf8be4f, BYTE-UENDRET. Co-Authored-By: Claude <claude-opus-5>
This commit is contained in:
parent
651c83f9df
commit
7aa06d581f
3 changed files with 513 additions and 7 deletions
380
tests/test_prepass_run_seam_loadbearing.py
Normal file
380
tests/test_prepass_run_seam_loadbearing.py
Normal file
|
|
@ -0,0 +1,380 @@
|
|||
"""Load-bearing gate for the pre-pass seam inside ``run_project`` (order 20260907T080223Z).
|
||||
|
||||
Given a verified payload, the bundle arm hands the debate a DECLARED CUT instead of
|
||||
``_bundle_pointer``'s pointer, and WITHDRAWS the four navigator tools. Without one, every byte of
|
||||
today's behaviour stands — which is the control every arm here is paired against.
|
||||
|
||||
**The withdrawal is asserted on the tool list passed to ``fresh_workflow``, never on
|
||||
``RunResult.debate_tool_calls``.** Measured: that trace is ALREADY empty with all four tools
|
||||
attached, because a ``ScriptedChatClient`` returns text and never emits a function call. An arm
|
||||
written on it cannot distinguish the two implementations at all.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
import portfolio_optimiser.run as run_module
|
||||
from portfolio_optimiser import okf, prepass
|
||||
from portfolio_optimiser.run import RunResult, run_project
|
||||
from portfolio_optimiser.mcp_tools import McpServerConfig
|
||||
from portfolio_optimiser.simulation import ScriptedChatClient
|
||||
from portfolio_optimiser.verdicts import VerdictStore, seed_store_from_bundle
|
||||
|
||||
FIXTURE = Path(__file__).parent / "fixtures" / "prepass" / "bygg-energi-mikro-fixture.payload.json"
|
||||
SHIPPED_BASE = Path(__file__).parent.parent / "shared" / "examples" / "bygg-energi-mikro"
|
||||
PROJECT_ID = "BYGG-KONTOR-NORD"
|
||||
NAVIGATOR_TOOLS = {"list_bundles", "read_bundle", "read_dir", "read_file"}
|
||||
|
||||
_LADDER = "read it with your tools"
|
||||
_EXCERPT_SENTINEL = "SENTINEL-I-ET-LEVERT-UTDRAG"
|
||||
|
||||
_PROPOSAL = json.dumps(
|
||||
{
|
||||
"project_id": PROJECT_ID,
|
||||
"measure": "energy_efficiency",
|
||||
"claimed_saving_nok": 30000,
|
||||
"affected_items": [
|
||||
{"code": "ENERGI-TOTAL-EL", "quantity": 120000.0, "unit_cost": 1.25},
|
||||
],
|
||||
"assumptions": {},
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _prompt_blob(messages: Any) -> str:
|
||||
"""Text PLUS function calls and results (the corrected S7a-2 instrument). ``.text`` alone
|
||||
measures a context-bearing prompt at a few characters."""
|
||||
parts: list[str] = []
|
||||
for message in messages:
|
||||
for content in getattr(message, "contents", []) or []:
|
||||
for attribute in ("text", "arguments", "result"):
|
||||
value = getattr(content, attribute, None)
|
||||
if value is not None:
|
||||
parts.append(str(value))
|
||||
return " ".join(parts)
|
||||
|
||||
|
||||
def _recording_factory(sink: list[str]) -> Any:
|
||||
"""A scripted client whose instance ``_inner_get_response`` is REBOUND to a recorder.
|
||||
|
||||
Rebinding rather than subclassing is deliberate: ``tests/test_scripted_client_consolidation``
|
||||
keeps a registry of every site that DEFINES that method, and a new definition here would go
|
||||
red there. This is the shape the two existing prompt gates use.
|
||||
"""
|
||||
|
||||
def factory(role: str) -> Any:
|
||||
client = ScriptedChatClient(_PROPOSAL, role=role)
|
||||
original = client._inner_get_response
|
||||
|
||||
async def recording(*args: Any, **kwargs: Any) -> Any:
|
||||
messages = kwargs.get("messages") or (args[0] if args else [])
|
||||
sink.append(_prompt_blob(messages))
|
||||
return await original(*args, **kwargs)
|
||||
|
||||
client._inner_get_response = recording # type: ignore[method-assign]
|
||||
return client
|
||||
|
||||
return factory
|
||||
|
||||
|
||||
def _base(tmp_path: Path, *, sentinel: bool = False) -> tuple[str, prepass.PrepassPayload]:
|
||||
"""A copy of the shipped base declaring its own id, with a matching payload.
|
||||
|
||||
The MOUNT differs from the DECLARATION (S7a-3's slack case), so the identity check cannot be
|
||||
satisfied by a mount comparison.
|
||||
"""
|
||||
root = tmp_path / "mounted-under-another-name"
|
||||
shutil.copytree(SHIPPED_BASE, root)
|
||||
index = root / "index.md"
|
||||
lines = index.read_text(encoding="utf-8").split("\n")
|
||||
lines.insert(1, "bundle_id: bygg-energi-mikro-fixture")
|
||||
index.write_text("\n".join(lines), encoding="utf-8")
|
||||
|
||||
raw = json.loads(FIXTURE.read_text(encoding="utf-8"))
|
||||
if sentinel:
|
||||
first = raw["excerpts"][0]
|
||||
path = root / (first["concept_id"] + ".md")
|
||||
path.write_text(
|
||||
path.read_text(encoding="utf-8") + f"\n\n{_EXCERPT_SENTINEL}\n", encoding="utf-8"
|
||||
)
|
||||
first["sha256"] = hashlib.sha256(path.read_bytes()).hexdigest()
|
||||
first["text"] = prepass.concept_text(path)
|
||||
first["text_sha256"] = hashlib.sha256(first["text"].encode("utf-8")).hexdigest()
|
||||
return str(root), prepass.PrepassPayload.model_validate(raw)
|
||||
|
||||
|
||||
def _docs(tmp_path: Path) -> str:
|
||||
"""The road path's data source, unused on the bundle arm but required by the signature."""
|
||||
d = tmp_path / "docs"
|
||||
d.mkdir(exist_ok=True)
|
||||
(d / "cost.txt").write_text("Energitiltak i kontorbygg.", encoding="utf-8")
|
||||
return str(d)
|
||||
|
||||
|
||||
async def _run(bundle_dir: str, **kwargs: Any) -> tuple[Any, list[str], list[list[Any]]]:
|
||||
"""Drive the bundle arm offline, capturing both the prompts and the tool list."""
|
||||
sink: list[str] = []
|
||||
captured: list[list[Any]] = []
|
||||
original = run_module.fresh_workflow
|
||||
|
||||
def spy(*args: Any, **spy_kwargs: Any) -> Any:
|
||||
captured.append(list(spy_kwargs.get("tools") or []))
|
||||
return original(*args, **spy_kwargs)
|
||||
|
||||
run_module.fresh_workflow = spy # type: ignore[assignment]
|
||||
try:
|
||||
result = await run_project(
|
||||
PROJECT_ID,
|
||||
bundle_dir=bundle_dir,
|
||||
docs_dir=_docs(Path(bundle_dir).parent),
|
||||
store=VerdictStore(verdicts=[]),
|
||||
client_factory=_recording_factory(sink),
|
||||
max_rounds=2,
|
||||
**kwargs,
|
||||
)
|
||||
finally:
|
||||
run_module.fresh_workflow = original # type: ignore[assignment]
|
||||
return result, sink, captured
|
||||
|
||||
|
||||
# --- the rendering replaces the pointer -----------------------------------------------------
|
||||
|
||||
|
||||
async def test_a_payload_replaces_the_pointer_in_the_debate(tmp_path: Path) -> None:
|
||||
bundle_dir, payload = _base(tmp_path, sentinel=True)
|
||||
_, sink, _ = await _run(bundle_dir, prepass_payload=payload)
|
||||
joined = " ".join(sink)
|
||||
assert _EXCERPT_SENTINEL in joined
|
||||
assert _LADDER not in joined
|
||||
|
||||
|
||||
async def test_without_a_payload_the_pointer_stands(tmp_path: Path) -> None:
|
||||
"""The control. Without it the arm above is satisfied by a run that built no prompt at all."""
|
||||
bundle_dir, _ = _base(tmp_path, sentinel=True)
|
||||
_, sink, _ = await _run(bundle_dir)
|
||||
joined = " ".join(sink)
|
||||
assert _LADDER in joined
|
||||
assert _EXCERPT_SENTINEL not in joined
|
||||
|
||||
|
||||
# --- the tools are withdrawn, and only they -------------------------------------------------
|
||||
|
||||
|
||||
async def test_a_payload_withdraws_the_four_navigator_tools(tmp_path: Path) -> None:
|
||||
bundle_dir, payload = _base(tmp_path)
|
||||
_, _, captured = await _run(bundle_dir, prepass_payload=payload)
|
||||
assert captured, "the debate was never built"
|
||||
assert {getattr(t, "name", "") for t in captured[0]} & NAVIGATOR_TOOLS == set()
|
||||
|
||||
|
||||
async def test_without_a_payload_the_debate_gets_exactly_those_four(tmp_path: Path) -> None:
|
||||
"""The paired control, asserting the EXACT set: `not any(...)` alone is also what an empty
|
||||
list produces, and an empty list is what a run that never built produces."""
|
||||
bundle_dir, _ = _base(tmp_path)
|
||||
_, _, captured = await _run(bundle_dir)
|
||||
assert captured, "the debate was never built"
|
||||
assert {getattr(t, "name", "") for t in captured[0]} == NAVIGATOR_TOOLS
|
||||
|
||||
|
||||
async def test_a_configured_mcp_tool_survives_the_withdrawal(tmp_path: Path) -> None:
|
||||
"""Distinguishes "withdraw the navigator tools" from ``debate_tools = []``. The MCP append
|
||||
sits BELOW the fork on purpose."""
|
||||
|
||||
class _FakeMcpTool:
|
||||
name = "an_external_tool"
|
||||
|
||||
async def __aenter__(self) -> "_FakeMcpTool":
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *exc: Any) -> None:
|
||||
return None
|
||||
|
||||
bundle_dir, payload = _base(tmp_path)
|
||||
original = run_module.build_mcp_tools
|
||||
run_module.build_mcp_tools = lambda servers: [_FakeMcpTool()] # type: ignore[assignment]
|
||||
try:
|
||||
_, _, captured = await _run(
|
||||
bundle_dir,
|
||||
prepass_payload=payload,
|
||||
mcp_servers=(
|
||||
McpServerConfig(
|
||||
name="prisregister",
|
||||
transport="http",
|
||||
url="https://intern.example/mcp",
|
||||
allowed_tools=("an_external_tool",),
|
||||
timeout_seconds=15,
|
||||
),
|
||||
),
|
||||
)
|
||||
finally:
|
||||
run_module.build_mcp_tools = original # type: ignore[assignment]
|
||||
names = {getattr(t, "name", "") for t in captured[0]}
|
||||
assert "an_external_tool" in names
|
||||
assert names & NAVIGATOR_TOOLS == set()
|
||||
|
||||
|
||||
# --- what else the fork must get right ------------------------------------------------------
|
||||
|
||||
|
||||
async def test_the_citations_are_the_delivered_concepts(tmp_path: Path) -> None:
|
||||
"""A stamp citing all five for a proposal that saw four re-creates the undeclared claim this
|
||||
seam removes. Each snippet stays exact by construction."""
|
||||
bundle_dir, payload = _base(tmp_path)
|
||||
result, _, _ = await _run(bundle_dir, prepass_payload=payload)
|
||||
assert isinstance(result, RunResult)
|
||||
cited = {c.file for c in result.provenance.citations}
|
||||
assert cited == {e.concept_id + ".md" for e in payload.excerpts}
|
||||
bodies = {f.name: f.body for f in okf.navigate_bundle(bundle_dir).context_files}
|
||||
for citation in result.provenance.citations:
|
||||
body = bodies[citation.file]
|
||||
assert citation.snippet == body[citation.locator.start_index : citation.locator.end_index]
|
||||
|
||||
|
||||
async def test_a_payload_run_reaches_an_outcome(tmp_path: Path) -> None:
|
||||
"""Nothing else here drives the fork past the debate; without this arm the citation guard at
|
||||
``run.py:1015``, the checker gate and the outbox path are all unexercised on the new path."""
|
||||
bundle_dir, payload = _base(tmp_path)
|
||||
result, _, _ = await _run(bundle_dir, prepass_payload=payload)
|
||||
assert isinstance(result, RunResult)
|
||||
assert result.provenance.validator_decision in {"validated", "rejected"}
|
||||
|
||||
|
||||
async def test_the_gated_expel_fold_still_reaches_the_hypothesis(tmp_path: Path) -> None:
|
||||
"""GATE, not wall. A prior judgement must still reach the hypothesis prompt through the
|
||||
ExpeL fold — otherwise an implementation that simply refuses everything verdict-shaped
|
||||
passes every other arm here while having removed the loop's whole learning path."""
|
||||
bundle_dir, payload = _base(tmp_path)
|
||||
store = seed_store_from_bundle(bundle_dir)
|
||||
assert store.verdicts, "the shipped base no longer seeds a verdict"
|
||||
|
||||
sink: list[str] = []
|
||||
await run_project(
|
||||
PROJECT_ID,
|
||||
bundle_dir=bundle_dir,
|
||||
docs_dir=_docs(Path(bundle_dir).parent),
|
||||
store=store,
|
||||
client_factory=_recording_factory(sink),
|
||||
max_rounds=2,
|
||||
prepass_payload=payload,
|
||||
)
|
||||
assert any("0.82" in prompt for prompt in sink), "the ExpeL fold no longer reaches generation"
|
||||
|
||||
|
||||
# --- refusals -------------------------------------------------------------------------------
|
||||
|
||||
|
||||
async def test_a_payload_for_another_base_refuses_without_building_a_debate(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
bundle_dir, _ = _base(tmp_path)
|
||||
raw = json.loads(FIXTURE.read_text(encoding="utf-8"))
|
||||
raw["bundle"]["bundle_id"] = "a-different-corpus"
|
||||
with pytest.raises(prepass.PrepassRefused):
|
||||
await _run(bundle_dir, prepass_payload=prepass.PrepassPayload.model_validate(raw))
|
||||
|
||||
|
||||
async def test_a_refused_payload_never_falls_back_to_the_pointer(tmp_path: Path) -> None:
|
||||
"""``load_mandate``'s rule: a caller who asked for a declared cut and got a navigating run
|
||||
instead was answered by a silently downgraded order."""
|
||||
bundle_dir, _ = _base(tmp_path)
|
||||
raw = json.loads(FIXTURE.read_text(encoding="utf-8"))
|
||||
raw["bundle"]["bundle_id"] = "a-different-corpus"
|
||||
sink: list[str] = []
|
||||
with pytest.raises(prepass.PrepassRefused):
|
||||
await run_project(
|
||||
PROJECT_ID,
|
||||
bundle_dir=bundle_dir,
|
||||
docs_dir=_docs(Path(bundle_dir).parent),
|
||||
store=VerdictStore(verdicts=[]),
|
||||
client_factory=_recording_factory(sink),
|
||||
max_rounds=2,
|
||||
prepass_payload=prepass.PrepassPayload.model_validate(raw),
|
||||
)
|
||||
assert sink == [], "the run made model calls despite refusing the payload"
|
||||
|
||||
|
||||
async def test_a_payload_that_delivers_nothing_is_refused_by_name(tmp_path: Path) -> None:
|
||||
"""Measured: ``delivered == 0`` is only reachable when every concept failed to match, which
|
||||
IS evidence of absence for this question at this ref. Saying so beats falling through to
|
||||
``run.py:1015``'s ``no citable content in docs_dir`` — a surface that is ``None`` here."""
|
||||
bundle_dir, _ = _base(tmp_path)
|
||||
raw = json.loads(FIXTURE.read_text(encoding="utf-8"))
|
||||
raw["withheld"] = [
|
||||
{"concept_id": e["concept_id"], "rule": "no_lexical_match"} for e in raw["excerpts"]
|
||||
] + raw["withheld"]
|
||||
raw["excerpts"] = []
|
||||
raw["denominators"]["withheld"] = len(raw["withheld"])
|
||||
raw["denominators"]["delivered"] = 0
|
||||
sink: list[str] = []
|
||||
with pytest.raises(prepass.PrepassRefused) as excinfo:
|
||||
await run_project(
|
||||
PROJECT_ID,
|
||||
bundle_dir=bundle_dir,
|
||||
docs_dir=_docs(Path(bundle_dir).parent),
|
||||
store=VerdictStore(verdicts=[]),
|
||||
client_factory=_recording_factory(sink),
|
||||
max_rounds=2,
|
||||
prepass_payload=prepass.PrepassPayload.model_validate(raw),
|
||||
)
|
||||
message = str(excinfo.value)
|
||||
assert "0" in message and raw["question"] in message
|
||||
assert sink == []
|
||||
|
||||
|
||||
async def test_a_payload_without_a_bundle_dir_is_refused(tmp_path: Path) -> None:
|
||||
"""The road path has no base for the payload to agree with."""
|
||||
_, payload = _base(tmp_path)
|
||||
with pytest.raises(prepass.PrepassRefused, match="bundle"):
|
||||
await run_project(
|
||||
PROJECT_ID,
|
||||
docs_dir=_docs(tmp_path),
|
||||
store=VerdictStore(verdicts=[]),
|
||||
client_factory=_recording_factory([]),
|
||||
max_rounds=2,
|
||||
prepass_payload=payload,
|
||||
)
|
||||
|
||||
|
||||
# --- the declaration on the result ----------------------------------------------------------
|
||||
|
||||
|
||||
async def test_the_run_carries_the_declaration(tmp_path: Path) -> None:
|
||||
bundle_dir, payload = _base(tmp_path)
|
||||
result, _, _ = await _run(bundle_dir, prepass_payload=payload)
|
||||
assert isinstance(result, RunResult)
|
||||
assert result.prepass is not None
|
||||
assert result.prepass.ref == payload.bundle.ref
|
||||
assert result.prepass.delivered == payload.denominators.delivered
|
||||
|
||||
|
||||
async def test_a_run_without_a_payload_carries_none(tmp_path: Path) -> None:
|
||||
bundle_dir, _ = _base(tmp_path)
|
||||
result, _, _ = await _run(bundle_dir)
|
||||
assert isinstance(result, RunResult)
|
||||
assert result.prepass is None
|
||||
|
||||
|
||||
async def test_the_dry_run_report_carries_the_declaration(tmp_path: Path) -> None:
|
||||
"""The dry-run cut sits BELOW the fork, so a dry run can honestly report the cut it was
|
||||
given — unlike ``unkeyed_verdicts``, which is resolved above it and could only report zero."""
|
||||
bundle_dir, payload = _base(tmp_path)
|
||||
report = await run_project(
|
||||
PROJECT_ID,
|
||||
bundle_dir=bundle_dir,
|
||||
docs_dir=_docs(Path(bundle_dir).parent),
|
||||
store=VerdictStore(verdicts=[]),
|
||||
client_factory=_recording_factory([]),
|
||||
max_rounds=2,
|
||||
live_dry_run=True,
|
||||
prepass_payload=payload,
|
||||
)
|
||||
assert isinstance(report, run_module.DryRunReport)
|
||||
assert report.prepass is not None
|
||||
assert report.prepass.considered == payload.denominators.considered
|
||||
Loading…
Add table
Add a link
Reference in a new issue