Three items on one seam — what a FAILED project does to the wave loop — plus the snapshot copy they sit next to. (v) The catch is BaseException, not Exception, and that width was ungated. The existing collect-and-continue test raises RuntimeError, so it stays green when the handler is narrowed: measured, the whole of test_portfolio_concurrent_ loadbearing.py (13 tests) passes under the narrowing. asyncio.CancelledError is the one realistic vector that separates the two — probed first, gather( return_exceptions=True) COLLECTS it, while KeyboardInterrupt propagates regardless and could never be helped by a wider catch. Narrowed, a cancelled member is cast into runs as a fake RunResult and the pass dies in _aggregate, pointing away from its cause. RED measured. (t) sum_token_usage excludes a failed project's spend, and that is the honest answer, not a bug: a run that died before producing a stamp has no provenance, and inventing one is the fabrication RunFailure exists to avoid. What needed gating is that those tokens still reach the ledger the global cap is enforced against — otherwise a repeatedly-failing project burns budget while the meter reads clean. Pins meter.spent as the pass's real cost, sum_token_usage as the completed-run subtotal, and their difference as exactly the failed spend. RED measured against the likely "fix" (sourcing sum_token_usage from the meter), which is wrong because a seeded meter also carries EARLIER passes' spend; 21 existing budget/portfolio tests stay green under it. (s) _wave_snapshot uses dataclasses.replace, so a field added later is carried without touching the function. Not cosmetic: measured, dropping retriever by hand-enumerating left all 585 tests green — the Step-2 coverage its docstring credited no longer existed, so the S3.1 retriever seam could be downgraded mid-pass in silence. Now gated by a property test derived from dataclasses.fields (not a field count, the shape rejected earlier). The explicit verdicts copy is retained and separately gated: replace(store) alone shares the caller's list and takes the byte-identical determinism test RED. strict=True on the zip is documented as deliberately untested — measured green when dropped, since gather is built from exactly snapshots, so a test could only go red by manufacturing a mismatch and would exercise zip rather than this pass. The new double is registered in the S2.5 consolidation guard's delegating- overrides list rather than the guard being weakened; it already delegates via super()._inner_get_response, which test_delegating_overrides_call_super now enforces on it. 583 -> 586 tests. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01MbgTCEZma764i1rHTrzceU
124 lines
6.3 KiB
Python
124 lines
6.3 KiB
Python
"""S2.5 (Step 9) consolidation guard (T-2.5e): the four scripted ``_inner_get_response`` bodies
|
|
collapsed to ONE canonical client (``simulation.ScriptedChatClient``); conftest's three test doubles
|
|
now SUBCLASS it. These grep-guards lock that in — they go RED if a divergent ``_inner_get_response``
|
|
is re-added, if a ``src``→``tests`` import creeps in, or if a double stops subclassing the canonical.
|
|
|
|
The guard is a regression lock over an already-verified consolidation: before the collapse there were
|
|
five ``def _inner_get_response`` sites (four scripted + test_step5's own-lineage double); after, two.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from pathlib import Path
|
|
|
|
_ROOT = Path(__file__).resolve().parents[1]
|
|
|
|
|
|
def _py_files(base: str) -> list[Path]:
|
|
return sorted((_ROOT / base).rglob("*.py"))
|
|
|
|
|
|
# Files permitted to define ``_inner_get_response``. The invariant this guard protects is that the
|
|
# scripted BODY is not duplicated — not that the def-site count is frozen. Adding a file here is a
|
|
# deliberate act: a new entry must either be the canonical, or a thin override that DELEGATES to it
|
|
# (which ``test_delegating_overrides_call_super`` below then enforces mechanically).
|
|
_CANONICAL_SITE = "src/portfolio_optimiser/simulation.py"
|
|
|
|
# Overrides in the SCRIPTED lineage — they subclass ``ScriptedChatClient``, so a body of their own
|
|
# would be a copy of the canonical. They must delegate.
|
|
_DELEGATING_OVERRIDES = [
|
|
# S3.3 ordering probe: yields to the event loop N times, then delegates. It cannot live in the
|
|
# reply-selector seam, which the canonical calls synchronously and so can never await.
|
|
"tests/test_portfolio_concurrent_loadbearing.py",
|
|
# S3.3 failure-accounting probe: RAISES for one project (after N completed calls), otherwise
|
|
# delegates. Like the ordering probe it cannot live in the reply-selector seam — that seam
|
|
# returns a reply string, and this double's whole subject is the absence of one.
|
|
"tests/test_portfolio_failure_accounting_loadbearing.py",
|
|
]
|
|
|
|
# Doubles in a DIFFERENT lineage (``spikes._harness.FakeChatClient``). There is no canonical
|
|
# scripted body above them to delegate to, so the delegation rule does not apply — but the
|
|
# separation is asserted rather than assumed, so a file cannot be parked here to dodge the rule.
|
|
_FOREIGN_LINEAGE = ["tests/test_step5_refine_loadbearing.py"]
|
|
|
|
|
|
def test_inner_get_response_collapsed_to_two_sites() -> None:
|
|
"""The four scripted clients collapse to ONE canonical ``_inner_get_response``
|
|
(``simulation.py``). Every other def-site must be a registered, DELEGATING override — never a
|
|
fourth copy of the body.
|
|
|
|
The guard originally pinned a literal count of 2. That made it fail on any new legitimate
|
|
subclass while still passing if someone pasted a duplicated body into an already-listed file —
|
|
a count is the wrong shape for the invariant. The list below plus
|
|
``test_delegating_overrides_call_super`` pin the property itself."""
|
|
sites = [
|
|
p.relative_to(_ROOT).as_posix()
|
|
for base in ("src", "tests")
|
|
for p in _py_files(base)
|
|
if p.name != Path(__file__).name # this guard file references the pattern in prose
|
|
and "def _inner_get_response" in p.read_text(encoding="utf-8")
|
|
]
|
|
expected = sorted([_CANONICAL_SITE, *_DELEGATING_OVERRIDES, *_FOREIGN_LINEAGE])
|
|
assert sorted(sites) == expected, (
|
|
f"unregistered ``_inner_get_response`` def-site — the scripted body must not be copied. "
|
|
f"Expected {expected}, got: {sites}"
|
|
)
|
|
|
|
|
|
def test_foreign_lineage_doubles_are_genuinely_foreign() -> None:
|
|
"""A file listed as foreign lineage must NOT subclass the scripted canonical.
|
|
|
|
Without this, ``_FOREIGN_LINEAGE`` would be an escape hatch: any scripted-lineage subclass
|
|
could be moved into that list to skip the delegation rule below."""
|
|
for site in _FOREIGN_LINEAGE:
|
|
text = (_ROOT / site).read_text(encoding="utf-8")
|
|
assert "ScriptedChatClient" not in text, (
|
|
f"{site} is registered as foreign lineage but references ``ScriptedChatClient`` — if it "
|
|
"is in the scripted lineage it belongs in _DELEGATING_OVERRIDES and must delegate"
|
|
)
|
|
|
|
|
|
def test_delegating_overrides_call_super() -> None:
|
|
"""Every scripted-lineage override actually DELEGATES to the canonical rather than
|
|
reimplementing it.
|
|
|
|
This is the strength the literal count never had: without it, a file already on the list could
|
|
grow a full copy of the scripted body and the consolidation would be cosmetic again."""
|
|
for site in _DELEGATING_OVERRIDES:
|
|
text = (_ROOT / site).read_text(encoding="utf-8")
|
|
# Match the delegation ITSELF — ``super()._inner_get_response`` or the explicit
|
|
# ``super(Cls, self)._inner_get_response`` form a nested function needs. Searching for
|
|
# "super(" and "_inner_get_response" independently would pass on any file that merely
|
|
# calls ``super().__init__`` near a def, which is accidental-green, not a guard.
|
|
assert re.search(r"super\([^)]*\)\._inner_get_response", text), (
|
|
f"{site} defines ``_inner_get_response`` but never delegates to the canonical via "
|
|
"``super()._inner_get_response`` — that is a duplicated body, which is exactly what "
|
|
"this guard exists to prevent"
|
|
)
|
|
|
|
|
|
def test_no_src_imports_tests() -> None:
|
|
"""No ``src`` module imports from ``tests`` — the canonical lives in ``src/simulation.py`` so
|
|
``conftest`` imports ``src``, never the reverse (the forbidden src→tests direction)."""
|
|
offenders = [
|
|
p.relative_to(_ROOT).as_posix()
|
|
for p in _py_files("src")
|
|
if ("from tests" in (text := p.read_text(encoding="utf-8")) or "import tests" in text)
|
|
]
|
|
assert offenders == [], f"src must not import tests: {offenders}"
|
|
|
|
|
|
def test_conftest_doubles_subclass_canonical() -> None:
|
|
"""conftest's three test doubles genuinely SUBCLASS the canonical ``ScriptedChatClient``
|
|
(delegating the shared body) — so the consolidation is real, not cosmetic."""
|
|
from conftest import (
|
|
ScriptedChatClient,
|
|
SyntheticUsageChatClient,
|
|
_ProjectAwareUsageChatClient,
|
|
_RecordingChatClient,
|
|
)
|
|
|
|
assert issubclass(SyntheticUsageChatClient, ScriptedChatClient)
|
|
assert issubclass(_ProjectAwareUsageChatClient, ScriptedChatClient)
|
|
assert issubclass(_RecordingChatClient, ScriptedChatClient)
|