test(loadbearing): positive controls for three measured-vacuous negatives
Continues the sibling-vacuity sweep (pkt. 2). Each fix is value-proved: GREEN BEFORE / RED AFTER under the same mutation, never a detach-proof alone. - test_goal_without_ledger_reads_an_empty_book asserted only `code == 0`. Measured: stubbing check_goal_before_spend to return False before ever reading the ledger left it GREEN — it could not tell "empty book, goal evaluated" from "check skipped", which is the one thing its name claims. Now asserts the evaluation line (realized 0 of 1.0 NOK, not reached). - test_the_allowlist_names_only_flags_the_cli_actually_has computed `missing == []` over _PORTFOLIO_SUPPORTED_DESTS. Measured: mutating the allowlist to frozenset() left it GREEN — an empty iteration yields an empty list, so a blind scanner reports no findings exactly as a clean one does. Now proves the detector fires on a flag the CLI lacks first. - test_the_system_prompt_is_empty asserted `system_prompt is None`, which is the SDK's OWN default (measured, 0.2.120). Deleting `system_prompt=None` from build_call_options left it GREEN: it pinned the SDK, not our code. The distinguishable seam is the Claude Code preset the S10 post-mortem retired, so the test now guards that and is renamed for what it proves. The None-vs-untouched limit is UNCONTROLLABLE and stated in the test. Also pins the SDK defaults both assertions choose against, so a future SDK shipping [] or a preset default degrades the anchor loudly instead of silently (an ANCHOR CAN DEGENERATE). Negative findings, so no session re-measures them: test_zero_model_calls is NOT vacuous (detaching the dry-run gate goes RED — though it dies inside the client, before reaching its own `calls == []`); notify/ir/validator/step1 and the two cli_paritet flag tests are covered by same-class sibling pairing on the same function. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01JQDNnD2szj3dthvqzd9Y8E
This commit is contained in:
parent
fae5b22578
commit
a43b5c7336
2 changed files with 72 additions and 9 deletions
|
|
@ -225,9 +225,17 @@ class TestGoalStopOnTheEntrance:
|
|||
assert "GOAL REACHED (soft)" in capsys.readouterr().out
|
||||
assert created and created[0].calls != []
|
||||
|
||||
def test_goal_without_ledger_reads_an_empty_book(self, tmp_path: Path) -> None:
|
||||
def test_goal_without_ledger_reads_an_empty_book(
|
||||
self, tmp_path: Path, capsys: pytest.CaptureFixture[str]
|
||||
) -> None:
|
||||
# An absent ledger is an EMPTY book (0 realized), never a skipped check:
|
||||
# the goal is evaluated, it is simply not reached.
|
||||
#
|
||||
# ``code == 0`` alone could NOT tell those two apart: measured, a
|
||||
# ``check_goal_before_spend`` stubbed to return False before ever reading
|
||||
# the ledger left this test green. The evaluation line is the only
|
||||
# observable that the book WAS read and scored as empty, so it is the
|
||||
# load-bearing assertion here — the exit code merely says nothing refused.
|
||||
factory, _ = _scripted_factory()
|
||||
code = main(
|
||||
[
|
||||
|
|
@ -241,6 +249,12 @@ class TestGoalStopOnTheEntrance:
|
|||
client_factory=factory,
|
||||
)
|
||||
assert code == 0
|
||||
out = capsys.readouterr().out
|
||||
assert "goal (hard): realized 0 of 1.0 NOK" in out, (
|
||||
"the goal was never evaluated against the empty book — an absent "
|
||||
"ledger must read as 0 realized, not as a skipped check"
|
||||
)
|
||||
assert "not reached" in out
|
||||
|
||||
def test_malformed_goal_is_refused_before_any_spend(self, tmp_path: Path) -> None:
|
||||
# §10: the goal contract is a startup contract — a percent goal is
|
||||
|
|
@ -490,12 +504,30 @@ class TestPortfolioClassifiesByAllowlist:
|
|||
# A renamed flag would leave a dead entry behind, and the renamed flag
|
||||
# would start being refused in portfolio mode without anyone saying so.
|
||||
help_text = _full_help("run")
|
||||
missing = sorted(
|
||||
f"--{dest.replace('_', '-')}"
|
||||
for dest in _PORTFOLIO_SUPPORTED_DESTS
|
||||
if f"--{dest.replace('_', '-')}" not in help_text
|
||||
|
||||
# Positive controls FIRST — the scan must have had something to scan and
|
||||
# must be able to report a miss. Measured: with the allowlist mutated to
|
||||
# ``frozenset()`` the ``missing == []`` below stayed green, because an
|
||||
# empty iteration yields an empty list. A blind scanner reports no
|
||||
# findings for the same reason a clean one does.
|
||||
assert _PORTFOLIO_SUPPORTED_DESTS, "the allowlist is empty — the check below is vacuous"
|
||||
assert "--portfolio" in help_text, "this does not look like `run --help`"
|
||||
absent = "tomorrows_flag"
|
||||
assert f"--{absent.replace('_', '-')}" not in help_text, (
|
||||
"the probe flag exists after all — pick one the CLI really lacks"
|
||||
)
|
||||
assert missing == []
|
||||
|
||||
def _missing_from_help(dests: frozenset[str]) -> list[str]:
|
||||
return sorted(
|
||||
f"--{dest.replace('_', '-')}"
|
||||
for dest in dests
|
||||
if f"--{dest.replace('_', '-')}" not in help_text
|
||||
)
|
||||
|
||||
# The detector demonstrably fires on a flag the CLI does not have...
|
||||
assert _missing_from_help(frozenset({absent})) == ["--tomorrows-flag"]
|
||||
# ...so an empty result for the real allowlist is a finding, not a no-op.
|
||||
assert _missing_from_help(_PORTFOLIO_SUPPORTED_DESTS) == []
|
||||
|
||||
|
||||
# --- the documentation-honesty seam (§1) -----------------------------------------------------
|
||||
|
|
|
|||
|
|
@ -17,7 +17,13 @@ from __future__ import annotations
|
|||
from typing import Any, AsyncIterator
|
||||
|
||||
import pytest
|
||||
from claude_agent_sdk import AssistantMessage, ResultMessage, TextBlock, ThinkingBlock
|
||||
from claude_agent_sdk import (
|
||||
AssistantMessage,
|
||||
ClaudeAgentOptions,
|
||||
ResultMessage,
|
||||
TextBlock,
|
||||
ThinkingBlock,
|
||||
)
|
||||
|
||||
from portfolio_optimiser_claude import sdk_client
|
||||
from portfolio_optimiser_claude.contracts import ModelMapContract
|
||||
|
|
@ -29,16 +35,41 @@ class TestBuildCallOptions:
|
|||
# LOAD-BEARING (§11): [] is SDK isolation mode. The default (None)
|
||||
# loads user+project+local settings — hooks and CLAUDE.md leak into
|
||||
# the model's context, exactly the observed S10 failure.
|
||||
#
|
||||
# Anchor-degeneration control: [] is only a CHOICE for as long as the
|
||||
# SDK's own default differs from it. Should a future SDK ship [] as
|
||||
# the default, the assertion below would keep passing while saying
|
||||
# nothing about our code — so pin the default we are choosing AGAINST.
|
||||
assert ClaudeAgentOptions().setting_sources is None, (
|
||||
"the SDK default changed — `setting_sources=[]` no longer proves an active choice"
|
||||
)
|
||||
options = build_call_options("model-x", max_budget_usd=0.25)
|
||||
assert options.setting_sources == []
|
||||
|
||||
def test_the_system_prompt_is_empty(self) -> None:
|
||||
# Pins the OPTION value: None, not the Claude Code preset. That None
|
||||
def test_the_system_prompt_is_not_the_claude_code_preset(self) -> None:
|
||||
# Pins the OPTION value against the Claude Code preset. That None
|
||||
# reaches the spawned CLI as --system-prompt "" was verified by
|
||||
# READING subprocess_cli.py (0.2.110–0.2.120) — this test does NOT
|
||||
# bind that transport serialization; doing so would couple the suite
|
||||
# to SDK-private API (the F11 fragility this repo retired).
|
||||
#
|
||||
# HONEST LIMIT (measured, 0.2.120): `system_prompt=None` is NOT
|
||||
# distinguishable from leaving the field untouched — the SDK default
|
||||
# is None too, so deleting `system_prompt=None` from build_call_options
|
||||
# left the old `is None` assertion GREEN. It pinned the SDK's default,
|
||||
# not our code. What IS load-bearing is the preset the S10 post-mortem
|
||||
# retired, so that is what this guards.
|
||||
assert ClaudeAgentOptions().system_prompt is None # the untouched-field baseline
|
||||
|
||||
# Positive control: the assertion below must be ABLE to fail. The
|
||||
# preset form is a real, accepted, distinguishable value.
|
||||
preset = ClaudeAgentOptions(
|
||||
model="model-x", system_prompt={"type": "preset", "preset": "claude_code"}
|
||||
)
|
||||
assert preset.system_prompt is not None
|
||||
|
||||
options = build_call_options("model-x", max_budget_usd=0.25)
|
||||
assert options.system_prompt != preset.system_prompt
|
||||
assert options.system_prompt is None
|
||||
|
||||
def test_the_call_stays_bounded_single_turn_no_tools(self) -> None:
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue