"""Prompt-isolation proof for the SDK call options (§11, S10 post-mortem). The S10 live run leaked the operator's Claude Code configuration into every spawned CLI session: ``ClaudeAgentOptions.setting_sources`` defaults to ``None``, which loads ALL filesystem settings (verified against SDK 0.2.110) — session-start hooks injected STATE.md into the model's context, every reply opened with a mandated confirmation line (so a reply was NEVER pure JSON), and each call paid ~10-15k uncached context tokens. ``[]`` is the SDK's documented isolation mode: no filesystem settings, no hooks, no CLAUDE.md. Importing the client here is offline-safe: constructing options touches no network and needs no API key; ``query()`` is replaced with a recording fake. """ from __future__ import annotations from typing import Any, AsyncIterator import pytest from claude_agent_sdk import AssistantMessage, ResultMessage, TextBlock, ThinkingBlock from portfolio_optimiser_claude import sdk_client from portfolio_optimiser_claude.contracts import ModelMapContract from portfolio_optimiser_claude.sdk_client import SdkModelClient, _total_tokens, build_call_options class TestBuildCallOptions: def test_all_filesystem_settings_are_disabled(self) -> None: # LOAD-BEARING (§11): [] is SDK isolation mode. The default (None) # loads user+project+local settings — hooks and CLAUDE.md leak into # the model's context, exactly the observed S10 failure. options = build_call_options("model-x", max_budget_usd=0.25) assert options.setting_sources == [] def test_the_system_prompt_is_empty(self) -> None: # Pins the OPTION value: None, not the Claude Code preset. That None # reaches the spawned CLI as --system-prompt "" was verified by # READING subprocess_cli.py (0.2.110–0.2.120) — this test does NOT # bind that transport serialization; doing so would couple the suite # to SDK-private API (the F11 fragility this repo retired). options = build_call_options("model-x", max_budget_usd=0.25) assert options.system_prompt is None def test_the_call_stays_bounded_single_turn_no_tools(self) -> None: options = build_call_options("model-x", max_budget_usd=0.25) assert options.max_turns == 1 assert options.tools == [] assert options.max_budget_usd == 0.25 assert options.model == "model-x" class TestCompleteThreadsIsolatedOptions: def test_complete_passes_the_isolated_options_to_query( self, monkeypatch: pytest.MonkeyPatch ) -> None: # Detach-proof: if complete() ever builds its options inline again # (dropping the isolation), this goes RED — grønn-men-død guard. captured: dict[str, Any] = {} def fake_query(*, prompt: str, options: Any) -> AsyncIterator[Any]: captured["prompt"] = prompt captured["options"] = options async def _empty() -> AsyncIterator[Any]: return yield return _empty() monkeypatch.setattr(sdk_client, "query", fake_query) client = SdkModelClient( ModelMapContract(profiles={"anthropic": {"default": "model-default"}}), max_budget_usd_per_call=0.10, ) reply = client.complete("the prompt", role="proposer") assert captured["prompt"] == "the prompt" assert captured["options"].setting_sources == [] assert captured["options"].max_budget_usd == 0.10 assert captured["options"].model == "model-default" # No usage surfaced by the fake → the reply fails CLOSED (§8). assert reply.usage_tokens is None def _stream_of(*messages: Any) -> Any: """A fake ``query`` yielding a scripted stream of REAL SDK message objects.""" def fake_query(*, prompt: str, options: Any) -> AsyncIterator[Any]: async def _stream() -> AsyncIterator[Any]: for message in messages: yield message return _stream() return fake_query def _assistant(*blocks: Any, model: str = "model-real", error: Any = None) -> AssistantMessage: return AssistantMessage(content=list(blocks), model=model, error=error) def _result( usage: dict[str, Any] | None = None, total_cost_usd: float | None = None, is_error: bool = False, subtype: str = "success", errors: list[str] | None = None, ) -> ResultMessage: return ResultMessage( subtype=subtype, duration_ms=1, duration_api_ms=1, is_error=is_error, num_turns=1, session_id="s", usage=usage, total_cost_usd=total_cost_usd, errors=errors, ) def _client() -> SdkModelClient: return SdkModelClient(ModelMapContract(profiles={"anthropic": {"default": "model-default"}})) _FULL_USAGE = { "input_tokens": 10, "output_tokens": 5, "cache_creation_input_tokens": 3, "cache_read_input_tokens": 2, } class TestCompleteAsyncStreamBinding: """C2.5 (R-4/R-5): the read loop is BOUND offline with real SDK message types. Before C2.5 nothing in the suite executed sdk_client's aggregation, error, usage or cost branches — the fake stream (real ``AssistantMessage`` / ``ResultMessage`` / ``TextBlock`` objects, so constructor drift also goes red) binds every branch without a key or the network. """ def test_text_aggregates_and_non_text_blocks_are_ignored( self, monkeypatch: pytest.MonkeyPatch ) -> None: monkeypatch.setattr( sdk_client, "query", _stream_of( _assistant(TextBlock("{"), ThinkingBlock(thinking="hmm", signature="sig")), _assistant(TextBlock("}")), _result(usage=_FULL_USAGE, total_cost_usd=0.01), ), ) client = _client() reply = client.complete("p", role="proposer") assert reply.text == "{}" assert reply.model == "model-real" assert client.last_model == "model-real" def test_usage_tokens_sum_the_four_provider_fields( self, monkeypatch: pytest.MonkeyPatch ) -> None: monkeypatch.setattr( sdk_client, "query", _stream_of(_assistant(TextBlock("ok")), _result(usage=_FULL_USAGE)), ) assert _client().complete("p", role="proposer").usage_tokens == 20 def test_cost_accumulates_across_calls(self, monkeypatch: pytest.MonkeyPatch) -> None: client = _client() for cost in (0.01, 0.02): monkeypatch.setattr( sdk_client, "query", _stream_of(_assistant(TextBlock("ok")), _result(total_cost_usd=cost)), ) client.complete("p", role="proposer") assert client.total_cost_usd == pytest.approx(0.03) def test_an_assistant_error_fails_the_call(self, monkeypatch: pytest.MonkeyPatch) -> None: monkeypatch.setattr( sdk_client, "query", _stream_of(_assistant(TextBlock("x"), error="rate_limit")) ) with pytest.raises(RuntimeError, match="rate_limit"): _client().complete("p", role="proposer") def test_a_result_error_fails_the_call_naming_subtype_and_errors( self, monkeypatch: pytest.MonkeyPatch ) -> None: monkeypatch.setattr( sdk_client, "query", _stream_of( _assistant(TextBlock("x")), _result(is_error=True, subtype="error_during_execution", errors=["boom"]), ), ) with pytest.raises(RuntimeError, match="error_during_execution.*boom"): _client().complete("p", role="proposer") class TestTotalTokensFailsClosed: """§8: the meter is never fed an invented count — no usage stays ``None``.""" def test_no_usage_dict_is_none(self) -> None: assert _total_tokens(None) is None def test_an_empty_usage_dict_is_none(self) -> None: assert _total_tokens({}) is None def test_non_int_fields_are_ignored_not_coerced(self) -> None: assert _total_tokens({"input_tokens": "10"}) is None assert _total_tokens({"input_tokens": 10, "output_tokens": "x"}) == 10 class TestBudgetGuard: """§8: a non-positive per-call USD cap is refused at construction.""" @pytest.mark.parametrize("cap", [0.0, -0.5]) def test_non_positive_caps_are_rejected(self, cap: float) -> None: with pytest.raises(ValueError): SdkModelClient( ModelMapContract(profiles={"anthropic": {"default": "m"}}), max_budget_usd_per_call=cap, )