feat(inbox): C2.5 — inbox hardening + SDK version guard (closes C-F7, C-N3, R-6)

- File-layer decision vocabulary (§4.2 set) with SKIP semantics — an unknown
  decision never reaches the store (C-F7, the review's run proof is the fixture)
- Fail-fast caps (max_files / max_rationale_chars) via InboxLimitError raised
  OUTSIDE the tolerant try — a cap breach is never swallowed as a skip
- R-6 id grammar (mirrors ingest _ID_RE) as a pydantic pattern on
  VerdictDocument.id AND re-checked in write_verdict, since model_copy(update=)
  bypasses model validation — traversal ids can no longer write outside the inbox
- promotion._filename_token: any sanitised id maps to a content hash — 'e/vil'
  can no longer clobber the distinct id 'evil' (restarbeid-funn 2)
- SDK pinned >=0.2.111,<0.3 + version guard test naming the sdk_client.py
  attribute premises; resolved 0.2.120, all premises re-verified against it
- sdk_client read loop bound offline with REAL SDK message types (R-4/R-5):
  text aggregation, error fail-paths, usage/cost extraction, _total_tokens
  fail-closed, non-positive budget guard
- test_sdk_isolation comment no longer claims the --system-prompt ""
  serialization the test body does not bind (honesty rule §1)

Guard-G2 assessment (guard-plan §4): the allowlist + caps + id grammar landed
here are G2's necessary part; an optional scan_output depth pass over
rationale (still a verbatim prose channel into the fold prompt, R-9) remains
relevant as a later additive session — the trigger picture is unchanged.

4 detach proofs red → restored green. Full gate: 389 passed (365→389),
ruff+format+mypy clean; golden + shared/ + runs/s10/ byte-untouched.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Kjell Tore Guttormsen 2026-07-16 20:26:41 +02:00
commit 80a2fa1a77
9 changed files with 441 additions and 35 deletions

View file

@ -17,10 +17,11 @@ from __future__ import annotations
from typing import Any, AsyncIterator
import pytest
from claude_agent_sdk import AssistantMessage, ResultMessage, TextBlock, ThinkingBlock
from portfolio_optimiser_claude import sdk_client
from portfolio_optimiser_claude.contracts import ModelMapContract
from portfolio_optimiser_claude.sdk_client import SdkModelClient, build_call_options
from portfolio_optimiser_claude.sdk_client import SdkModelClient, _total_tokens, build_call_options
class TestBuildCallOptions:
@ -32,8 +33,11 @@ class TestBuildCallOptions:
assert options.setting_sources == []
def test_the_system_prompt_is_empty(self) -> None:
# None serializes to --system-prompt "" (verified against 0.2.110):
# no Claude Code preset, no appended operator instructions.
# Pins the OPTION value: None, not the Claude Code preset. That None
# reaches the spawned CLI as --system-prompt "" was verified by
# READING subprocess_cli.py (0.2.1100.2.120) — this test does NOT
# bind that transport serialization; doing so would couple the suite
# to SDK-private API (the F11 fragility this repo retired).
options = build_call_options("model-x", max_budget_usd=0.25)
assert options.system_prompt is None
@ -75,3 +79,148 @@ class TestCompleteThreadsIsolatedOptions:
assert captured["options"].model == "model-default"
# No usage surfaced by the fake → the reply fails CLOSED (§8).
assert reply.usage_tokens is None
def _stream_of(*messages: Any) -> Any:
"""A fake ``query`` yielding a scripted stream of REAL SDK message objects."""
def fake_query(*, prompt: str, options: Any) -> AsyncIterator[Any]:
async def _stream() -> AsyncIterator[Any]:
for message in messages:
yield message
return _stream()
return fake_query
def _assistant(*blocks: Any, model: str = "model-real", error: Any = None) -> AssistantMessage:
return AssistantMessage(content=list(blocks), model=model, error=error)
def _result(
usage: dict[str, Any] | None = None,
total_cost_usd: float | None = None,
is_error: bool = False,
subtype: str = "success",
errors: list[str] | None = None,
) -> ResultMessage:
return ResultMessage(
subtype=subtype,
duration_ms=1,
duration_api_ms=1,
is_error=is_error,
num_turns=1,
session_id="s",
usage=usage,
total_cost_usd=total_cost_usd,
errors=errors,
)
def _client() -> SdkModelClient:
return SdkModelClient(ModelMapContract(profiles={"anthropic": {"default": "model-default"}}))
_FULL_USAGE = {
"input_tokens": 10,
"output_tokens": 5,
"cache_creation_input_tokens": 3,
"cache_read_input_tokens": 2,
}
class TestCompleteAsyncStreamBinding:
"""C2.5 (R-4/R-5): the read loop is BOUND offline with real SDK message types.
Before C2.5 nothing in the suite executed sdk_client's aggregation,
error, usage or cost branches the fake stream (real ``AssistantMessage``
/ ``ResultMessage`` / ``TextBlock`` objects, so constructor drift also
goes red) binds every branch without a key or the network.
"""
def test_text_aggregates_and_non_text_blocks_are_ignored(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(
sdk_client,
"query",
_stream_of(
_assistant(TextBlock("{"), ThinkingBlock(thinking="hmm", signature="sig")),
_assistant(TextBlock("}")),
_result(usage=_FULL_USAGE, total_cost_usd=0.01),
),
)
client = _client()
reply = client.complete("p", role="proposer")
assert reply.text == "{}"
assert reply.model == "model-real"
assert client.last_model == "model-real"
def test_usage_tokens_sum_the_four_provider_fields(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(
sdk_client,
"query",
_stream_of(_assistant(TextBlock("ok")), _result(usage=_FULL_USAGE)),
)
assert _client().complete("p", role="proposer").usage_tokens == 20
def test_cost_accumulates_across_calls(self, monkeypatch: pytest.MonkeyPatch) -> None:
client = _client()
for cost in (0.01, 0.02):
monkeypatch.setattr(
sdk_client,
"query",
_stream_of(_assistant(TextBlock("ok")), _result(total_cost_usd=cost)),
)
client.complete("p", role="proposer")
assert client.total_cost_usd == pytest.approx(0.03)
def test_an_assistant_error_fails_the_call(self, monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(
sdk_client, "query", _stream_of(_assistant(TextBlock("x"), error="rate_limit"))
)
with pytest.raises(RuntimeError, match="rate_limit"):
_client().complete("p", role="proposer")
def test_a_result_error_fails_the_call_naming_subtype_and_errors(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(
sdk_client,
"query",
_stream_of(
_assistant(TextBlock("x")),
_result(is_error=True, subtype="error_during_execution", errors=["boom"]),
),
)
with pytest.raises(RuntimeError, match="error_during_execution.*boom"):
_client().complete("p", role="proposer")
class TestTotalTokensFailsClosed:
"""§8: the meter is never fed an invented count — no usage stays ``None``."""
def test_no_usage_dict_is_none(self) -> None:
assert _total_tokens(None) is None
def test_an_empty_usage_dict_is_none(self) -> None:
assert _total_tokens({}) is None
def test_non_int_fields_are_ignored_not_coerced(self) -> None:
assert _total_tokens({"input_tokens": "10"}) is None
assert _total_tokens({"input_tokens": 10, "output_tokens": "x"}) == 10
class TestBudgetGuard:
"""§8: a non-positive per-call USD cap is refused at construction."""
@pytest.mark.parametrize("cap", [0.0, -0.5])
def test_non_positive_caps_are_rejected(self, cap: float) -> None:
with pytest.raises(ValueError):
SdkModelClient(
ModelMapContract(profiles={"anthropic": {"default": "m"}}),
max_budget_usd_per_call=cap,
)