- File-layer decision vocabulary (§4.2 set) with SKIP semantics — an unknown decision never reaches the store (C-F7, the review's run proof is the fixture) - Fail-fast caps (max_files / max_rationale_chars) via InboxLimitError raised OUTSIDE the tolerant try — a cap breach is never swallowed as a skip - R-6 id grammar (mirrors ingest _ID_RE) as a pydantic pattern on VerdictDocument.id AND re-checked in write_verdict, since model_copy(update=) bypasses model validation — traversal ids can no longer write outside the inbox - promotion._filename_token: any sanitised id maps to a content hash — 'e/vil' can no longer clobber the distinct id 'evil' (restarbeid-funn 2) - SDK pinned >=0.2.111,<0.3 + version guard test naming the sdk_client.py attribute premises; resolved 0.2.120, all premises re-verified against it - sdk_client read loop bound offline with REAL SDK message types (R-4/R-5): text aggregation, error fail-paths, usage/cost extraction, _total_tokens fail-closed, non-positive budget guard - test_sdk_isolation comment no longer claims the --system-prompt "" serialization the test body does not bind (honesty rule §1) Guard-G2 assessment (guard-plan §4): the allowlist + caps + id grammar landed here are G2's necessary part; an optional scan_output depth pass over rationale (still a verbatim prose channel into the fold prompt, R-9) remains relevant as a later additive session — the trigger picture is unchanged. 4 detach proofs red → restored green. Full gate: 389 passed (365→389), ruff+format+mypy clean; golden + shared/ + runs/s10/ byte-untouched. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
226 lines
8.5 KiB
Python
226 lines
8.5 KiB
Python
"""Prompt-isolation proof for the SDK call options (§11, S10 post-mortem).
|
||
|
||
The S10 live run leaked the operator's Claude Code configuration into every
|
||
spawned CLI session: ``ClaudeAgentOptions.setting_sources`` defaults to
|
||
``None``, which loads ALL filesystem settings (verified against SDK 0.2.110)
|
||
— session-start hooks injected STATE.md into the model's context, every reply
|
||
opened with a mandated confirmation line (so a reply was NEVER pure JSON),
|
||
and each call paid ~10-15k uncached context tokens. ``[]`` is the SDK's
|
||
documented isolation mode: no filesystem settings, no hooks, no CLAUDE.md.
|
||
|
||
Importing the client here is offline-safe: constructing options touches no
|
||
network and needs no API key; ``query()`` is replaced with a recording fake.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from typing import Any, AsyncIterator
|
||
|
||
import pytest
|
||
from claude_agent_sdk import AssistantMessage, ResultMessage, TextBlock, ThinkingBlock
|
||
|
||
from portfolio_optimiser_claude import sdk_client
|
||
from portfolio_optimiser_claude.contracts import ModelMapContract
|
||
from portfolio_optimiser_claude.sdk_client import SdkModelClient, _total_tokens, build_call_options
|
||
|
||
|
||
class TestBuildCallOptions:
|
||
def test_all_filesystem_settings_are_disabled(self) -> None:
|
||
# LOAD-BEARING (§11): [] is SDK isolation mode. The default (None)
|
||
# loads user+project+local settings — hooks and CLAUDE.md leak into
|
||
# the model's context, exactly the observed S10 failure.
|
||
options = build_call_options("model-x", max_budget_usd=0.25)
|
||
assert options.setting_sources == []
|
||
|
||
def test_the_system_prompt_is_empty(self) -> None:
|
||
# Pins the OPTION value: None, not the Claude Code preset. That None
|
||
# reaches the spawned CLI as --system-prompt "" was verified by
|
||
# READING subprocess_cli.py (0.2.110–0.2.120) — this test does NOT
|
||
# bind that transport serialization; doing so would couple the suite
|
||
# to SDK-private API (the F11 fragility this repo retired).
|
||
options = build_call_options("model-x", max_budget_usd=0.25)
|
||
assert options.system_prompt is None
|
||
|
||
def test_the_call_stays_bounded_single_turn_no_tools(self) -> None:
|
||
options = build_call_options("model-x", max_budget_usd=0.25)
|
||
assert options.max_turns == 1
|
||
assert options.tools == []
|
||
assert options.max_budget_usd == 0.25
|
||
assert options.model == "model-x"
|
||
|
||
|
||
class TestCompleteThreadsIsolatedOptions:
|
||
def test_complete_passes_the_isolated_options_to_query(
|
||
self, monkeypatch: pytest.MonkeyPatch
|
||
) -> None:
|
||
# Detach-proof: if complete() ever builds its options inline again
|
||
# (dropping the isolation), this goes RED — grønn-men-død guard.
|
||
captured: dict[str, Any] = {}
|
||
|
||
def fake_query(*, prompt: str, options: Any) -> AsyncIterator[Any]:
|
||
captured["prompt"] = prompt
|
||
captured["options"] = options
|
||
|
||
async def _empty() -> AsyncIterator[Any]:
|
||
return
|
||
yield
|
||
|
||
return _empty()
|
||
|
||
monkeypatch.setattr(sdk_client, "query", fake_query)
|
||
client = SdkModelClient(
|
||
ModelMapContract(profiles={"anthropic": {"default": "model-default"}}),
|
||
max_budget_usd_per_call=0.10,
|
||
)
|
||
reply = client.complete("the prompt", role="proposer")
|
||
assert captured["prompt"] == "the prompt"
|
||
assert captured["options"].setting_sources == []
|
||
assert captured["options"].max_budget_usd == 0.10
|
||
assert captured["options"].model == "model-default"
|
||
# No usage surfaced by the fake → the reply fails CLOSED (§8).
|
||
assert reply.usage_tokens is None
|
||
|
||
|
||
def _stream_of(*messages: Any) -> Any:
|
||
"""A fake ``query`` yielding a scripted stream of REAL SDK message objects."""
|
||
|
||
def fake_query(*, prompt: str, options: Any) -> AsyncIterator[Any]:
|
||
async def _stream() -> AsyncIterator[Any]:
|
||
for message in messages:
|
||
yield message
|
||
|
||
return _stream()
|
||
|
||
return fake_query
|
||
|
||
|
||
def _assistant(*blocks: Any, model: str = "model-real", error: Any = None) -> AssistantMessage:
|
||
return AssistantMessage(content=list(blocks), model=model, error=error)
|
||
|
||
|
||
def _result(
|
||
usage: dict[str, Any] | None = None,
|
||
total_cost_usd: float | None = None,
|
||
is_error: bool = False,
|
||
subtype: str = "success",
|
||
errors: list[str] | None = None,
|
||
) -> ResultMessage:
|
||
return ResultMessage(
|
||
subtype=subtype,
|
||
duration_ms=1,
|
||
duration_api_ms=1,
|
||
is_error=is_error,
|
||
num_turns=1,
|
||
session_id="s",
|
||
usage=usage,
|
||
total_cost_usd=total_cost_usd,
|
||
errors=errors,
|
||
)
|
||
|
||
|
||
def _client() -> SdkModelClient:
|
||
return SdkModelClient(ModelMapContract(profiles={"anthropic": {"default": "model-default"}}))
|
||
|
||
|
||
_FULL_USAGE = {
|
||
"input_tokens": 10,
|
||
"output_tokens": 5,
|
||
"cache_creation_input_tokens": 3,
|
||
"cache_read_input_tokens": 2,
|
||
}
|
||
|
||
|
||
class TestCompleteAsyncStreamBinding:
|
||
"""C2.5 (R-4/R-5): the read loop is BOUND offline with real SDK message types.
|
||
|
||
Before C2.5 nothing in the suite executed sdk_client's aggregation,
|
||
error, usage or cost branches — the fake stream (real ``AssistantMessage``
|
||
/ ``ResultMessage`` / ``TextBlock`` objects, so constructor drift also
|
||
goes red) binds every branch without a key or the network.
|
||
"""
|
||
|
||
def test_text_aggregates_and_non_text_blocks_are_ignored(
|
||
self, monkeypatch: pytest.MonkeyPatch
|
||
) -> None:
|
||
monkeypatch.setattr(
|
||
sdk_client,
|
||
"query",
|
||
_stream_of(
|
||
_assistant(TextBlock("{"), ThinkingBlock(thinking="hmm", signature="sig")),
|
||
_assistant(TextBlock("}")),
|
||
_result(usage=_FULL_USAGE, total_cost_usd=0.01),
|
||
),
|
||
)
|
||
client = _client()
|
||
reply = client.complete("p", role="proposer")
|
||
assert reply.text == "{}"
|
||
assert reply.model == "model-real"
|
||
assert client.last_model == "model-real"
|
||
|
||
def test_usage_tokens_sum_the_four_provider_fields(
|
||
self, monkeypatch: pytest.MonkeyPatch
|
||
) -> None:
|
||
monkeypatch.setattr(
|
||
sdk_client,
|
||
"query",
|
||
_stream_of(_assistant(TextBlock("ok")), _result(usage=_FULL_USAGE)),
|
||
)
|
||
assert _client().complete("p", role="proposer").usage_tokens == 20
|
||
|
||
def test_cost_accumulates_across_calls(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
||
client = _client()
|
||
for cost in (0.01, 0.02):
|
||
monkeypatch.setattr(
|
||
sdk_client,
|
||
"query",
|
||
_stream_of(_assistant(TextBlock("ok")), _result(total_cost_usd=cost)),
|
||
)
|
||
client.complete("p", role="proposer")
|
||
assert client.total_cost_usd == pytest.approx(0.03)
|
||
|
||
def test_an_assistant_error_fails_the_call(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
||
monkeypatch.setattr(
|
||
sdk_client, "query", _stream_of(_assistant(TextBlock("x"), error="rate_limit"))
|
||
)
|
||
with pytest.raises(RuntimeError, match="rate_limit"):
|
||
_client().complete("p", role="proposer")
|
||
|
||
def test_a_result_error_fails_the_call_naming_subtype_and_errors(
|
||
self, monkeypatch: pytest.MonkeyPatch
|
||
) -> None:
|
||
monkeypatch.setattr(
|
||
sdk_client,
|
||
"query",
|
||
_stream_of(
|
||
_assistant(TextBlock("x")),
|
||
_result(is_error=True, subtype="error_during_execution", errors=["boom"]),
|
||
),
|
||
)
|
||
with pytest.raises(RuntimeError, match="error_during_execution.*boom"):
|
||
_client().complete("p", role="proposer")
|
||
|
||
|
||
class TestTotalTokensFailsClosed:
|
||
"""§8: the meter is never fed an invented count — no usage stays ``None``."""
|
||
|
||
def test_no_usage_dict_is_none(self) -> None:
|
||
assert _total_tokens(None) is None
|
||
|
||
def test_an_empty_usage_dict_is_none(self) -> None:
|
||
assert _total_tokens({}) is None
|
||
|
||
def test_non_int_fields_are_ignored_not_coerced(self) -> None:
|
||
assert _total_tokens({"input_tokens": "10"}) is None
|
||
assert _total_tokens({"input_tokens": 10, "output_tokens": "x"}) == 10
|
||
|
||
|
||
class TestBudgetGuard:
|
||
"""§8: a non-positive per-call USD cap is refused at construction."""
|
||
|
||
@pytest.mark.parametrize("cap", [0.0, -0.5])
|
||
def test_non_positive_caps_are_rejected(self, cap: float) -> None:
|
||
with pytest.raises(ValueError):
|
||
SdkModelClient(
|
||
ModelMapContract(profiles={"anthropic": {"default": "m"}}),
|
||
max_budget_usd_per_call=cap,
|
||
)
|