portfolio-optimiser-claude/tests/test_sdk_isolation.py
Kjell Tore Guttormsen 80a2fa1a77 feat(inbox): C2.5 — inbox hardening + SDK version guard (closes C-F7, C-N3, R-6)
- File-layer decision vocabulary (§4.2 set) with SKIP semantics — an unknown
  decision never reaches the store (C-F7, the review's run proof is the fixture)
- Fail-fast caps (max_files / max_rationale_chars) via InboxLimitError raised
  OUTSIDE the tolerant try — a cap breach is never swallowed as a skip
- R-6 id grammar (mirrors ingest _ID_RE) as a pydantic pattern on
  VerdictDocument.id AND re-checked in write_verdict, since model_copy(update=)
  bypasses model validation — traversal ids can no longer write outside the inbox
- promotion._filename_token: any sanitised id maps to a content hash — 'e/vil'
  can no longer clobber the distinct id 'evil' (restarbeid-funn 2)
- SDK pinned >=0.2.111,<0.3 + version guard test naming the sdk_client.py
  attribute premises; resolved 0.2.120, all premises re-verified against it
- sdk_client read loop bound offline with REAL SDK message types (R-4/R-5):
  text aggregation, error fail-paths, usage/cost extraction, _total_tokens
  fail-closed, non-positive budget guard
- test_sdk_isolation comment no longer claims the --system-prompt ""
  serialization the test body does not bind (honesty rule §1)

Guard-G2 assessment (guard-plan §4): the allowlist + caps + id grammar landed
here are G2's necessary part; an optional scan_output depth pass over
rationale (still a verbatim prose channel into the fold prompt, R-9) remains
relevant as a later additive session — the trigger picture is unchanged.

4 detach proofs red → restored green. Full gate: 389 passed (365→389),
ruff+format+mypy clean; golden + shared/ + runs/s10/ byte-untouched.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-16 20:26:41 +02:00

226 lines
8.5 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Prompt-isolation proof for the SDK call options (§11, S10 post-mortem).
The S10 live run leaked the operator's Claude Code configuration into every
spawned CLI session: ``ClaudeAgentOptions.setting_sources`` defaults to
``None``, which loads ALL filesystem settings (verified against SDK 0.2.110)
— session-start hooks injected STATE.md into the model's context, every reply
opened with a mandated confirmation line (so a reply was NEVER pure JSON),
and each call paid ~10-15k uncached context tokens. ``[]`` is the SDK's
documented isolation mode: no filesystem settings, no hooks, no CLAUDE.md.
Importing the client here is offline-safe: constructing options touches no
network and needs no API key; ``query()`` is replaced with a recording fake.
"""
from __future__ import annotations
from typing import Any, AsyncIterator
import pytest
from claude_agent_sdk import AssistantMessage, ResultMessage, TextBlock, ThinkingBlock
from portfolio_optimiser_claude import sdk_client
from portfolio_optimiser_claude.contracts import ModelMapContract
from portfolio_optimiser_claude.sdk_client import SdkModelClient, _total_tokens, build_call_options
class TestBuildCallOptions:
def test_all_filesystem_settings_are_disabled(self) -> None:
# LOAD-BEARING (§11): [] is SDK isolation mode. The default (None)
# loads user+project+local settings — hooks and CLAUDE.md leak into
# the model's context, exactly the observed S10 failure.
options = build_call_options("model-x", max_budget_usd=0.25)
assert options.setting_sources == []
def test_the_system_prompt_is_empty(self) -> None:
# Pins the OPTION value: None, not the Claude Code preset. That None
# reaches the spawned CLI as --system-prompt "" was verified by
# READING subprocess_cli.py (0.2.1100.2.120) — this test does NOT
# bind that transport serialization; doing so would couple the suite
# to SDK-private API (the F11 fragility this repo retired).
options = build_call_options("model-x", max_budget_usd=0.25)
assert options.system_prompt is None
def test_the_call_stays_bounded_single_turn_no_tools(self) -> None:
options = build_call_options("model-x", max_budget_usd=0.25)
assert options.max_turns == 1
assert options.tools == []
assert options.max_budget_usd == 0.25
assert options.model == "model-x"
class TestCompleteThreadsIsolatedOptions:
def test_complete_passes_the_isolated_options_to_query(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
# Detach-proof: if complete() ever builds its options inline again
# (dropping the isolation), this goes RED — grønn-men-død guard.
captured: dict[str, Any] = {}
def fake_query(*, prompt: str, options: Any) -> AsyncIterator[Any]:
captured["prompt"] = prompt
captured["options"] = options
async def _empty() -> AsyncIterator[Any]:
return
yield
return _empty()
monkeypatch.setattr(sdk_client, "query", fake_query)
client = SdkModelClient(
ModelMapContract(profiles={"anthropic": {"default": "model-default"}}),
max_budget_usd_per_call=0.10,
)
reply = client.complete("the prompt", role="proposer")
assert captured["prompt"] == "the prompt"
assert captured["options"].setting_sources == []
assert captured["options"].max_budget_usd == 0.10
assert captured["options"].model == "model-default"
# No usage surfaced by the fake → the reply fails CLOSED (§8).
assert reply.usage_tokens is None
def _stream_of(*messages: Any) -> Any:
"""A fake ``query`` yielding a scripted stream of REAL SDK message objects."""
def fake_query(*, prompt: str, options: Any) -> AsyncIterator[Any]:
async def _stream() -> AsyncIterator[Any]:
for message in messages:
yield message
return _stream()
return fake_query
def _assistant(*blocks: Any, model: str = "model-real", error: Any = None) -> AssistantMessage:
return AssistantMessage(content=list(blocks), model=model, error=error)
def _result(
usage: dict[str, Any] | None = None,
total_cost_usd: float | None = None,
is_error: bool = False,
subtype: str = "success",
errors: list[str] | None = None,
) -> ResultMessage:
return ResultMessage(
subtype=subtype,
duration_ms=1,
duration_api_ms=1,
is_error=is_error,
num_turns=1,
session_id="s",
usage=usage,
total_cost_usd=total_cost_usd,
errors=errors,
)
def _client() -> SdkModelClient:
return SdkModelClient(ModelMapContract(profiles={"anthropic": {"default": "model-default"}}))
_FULL_USAGE = {
"input_tokens": 10,
"output_tokens": 5,
"cache_creation_input_tokens": 3,
"cache_read_input_tokens": 2,
}
class TestCompleteAsyncStreamBinding:
"""C2.5 (R-4/R-5): the read loop is BOUND offline with real SDK message types.
Before C2.5 nothing in the suite executed sdk_client's aggregation,
error, usage or cost branches — the fake stream (real ``AssistantMessage``
/ ``ResultMessage`` / ``TextBlock`` objects, so constructor drift also
goes red) binds every branch without a key or the network.
"""
def test_text_aggregates_and_non_text_blocks_are_ignored(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(
sdk_client,
"query",
_stream_of(
_assistant(TextBlock("{"), ThinkingBlock(thinking="hmm", signature="sig")),
_assistant(TextBlock("}")),
_result(usage=_FULL_USAGE, total_cost_usd=0.01),
),
)
client = _client()
reply = client.complete("p", role="proposer")
assert reply.text == "{}"
assert reply.model == "model-real"
assert client.last_model == "model-real"
def test_usage_tokens_sum_the_four_provider_fields(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(
sdk_client,
"query",
_stream_of(_assistant(TextBlock("ok")), _result(usage=_FULL_USAGE)),
)
assert _client().complete("p", role="proposer").usage_tokens == 20
def test_cost_accumulates_across_calls(self, monkeypatch: pytest.MonkeyPatch) -> None:
client = _client()
for cost in (0.01, 0.02):
monkeypatch.setattr(
sdk_client,
"query",
_stream_of(_assistant(TextBlock("ok")), _result(total_cost_usd=cost)),
)
client.complete("p", role="proposer")
assert client.total_cost_usd == pytest.approx(0.03)
def test_an_assistant_error_fails_the_call(self, monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(
sdk_client, "query", _stream_of(_assistant(TextBlock("x"), error="rate_limit"))
)
with pytest.raises(RuntimeError, match="rate_limit"):
_client().complete("p", role="proposer")
def test_a_result_error_fails_the_call_naming_subtype_and_errors(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(
sdk_client,
"query",
_stream_of(
_assistant(TextBlock("x")),
_result(is_error=True, subtype="error_during_execution", errors=["boom"]),
),
)
with pytest.raises(RuntimeError, match="error_during_execution.*boom"):
_client().complete("p", role="proposer")
class TestTotalTokensFailsClosed:
"""§8: the meter is never fed an invented count — no usage stays ``None``."""
def test_no_usage_dict_is_none(self) -> None:
assert _total_tokens(None) is None
def test_an_empty_usage_dict_is_none(self) -> None:
assert _total_tokens({}) is None
def test_non_int_fields_are_ignored_not_coerced(self) -> None:
assert _total_tokens({"input_tokens": "10"}) is None
assert _total_tokens({"input_tokens": 10, "output_tokens": "x"}) == 10
class TestBudgetGuard:
"""§8: a non-positive per-call USD cap is refused at construction."""
@pytest.mark.parametrize("cap", [0.0, -0.5])
def test_non_positive_caps_are_rejected(self, cap: float) -> None:
with pytest.raises(ValueError):
SdkModelClient(
ModelMapContract(profiles={"anthropic": {"default": "m"}}),
max_budget_usd_per_call=cap,
)