feat(inbox): C2.5 — inbox hardening + SDK version guard (closes C-F7, C-N3, R-6)
- File-layer decision vocabulary (§4.2 set) with SKIP semantics — an unknown decision never reaches the store (C-F7, the review's run proof is the fixture) - Fail-fast caps (max_files / max_rationale_chars) via InboxLimitError raised OUTSIDE the tolerant try — a cap breach is never swallowed as a skip - R-6 id grammar (mirrors ingest _ID_RE) as a pydantic pattern on VerdictDocument.id AND re-checked in write_verdict, since model_copy(update=) bypasses model validation — traversal ids can no longer write outside the inbox - promotion._filename_token: any sanitised id maps to a content hash — 'e/vil' can no longer clobber the distinct id 'evil' (restarbeid-funn 2) - SDK pinned >=0.2.111,<0.3 + version guard test naming the sdk_client.py attribute premises; resolved 0.2.120, all premises re-verified against it - sdk_client read loop bound offline with REAL SDK message types (R-4/R-5): text aggregation, error fail-paths, usage/cost extraction, _total_tokens fail-closed, non-positive budget guard - test_sdk_isolation comment no longer claims the --system-prompt "" serialization the test body does not bind (honesty rule §1) Guard-G2 assessment (guard-plan §4): the allowlist + caps + id grammar landed here are G2's necessary part; an optional scan_output depth pass over rationale (still a verbatim prose channel into the fold prompt, R-9) remains relevant as a later additive session — the trigger picture is unchanged. 4 detach proofs red → restored green. Full gate: 389 passed (365→389), ruff+format+mypy clean; golden + shared/ + runs/s10/ byte-untouched. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
parent
e7ce6b0a31
commit
80a2fa1a77
9 changed files with 441 additions and 35 deletions
|
|
@ -17,10 +17,11 @@ from __future__ import annotations
|
|||
from typing import Any, AsyncIterator
|
||||
|
||||
import pytest
|
||||
from claude_agent_sdk import AssistantMessage, ResultMessage, TextBlock, ThinkingBlock
|
||||
|
||||
from portfolio_optimiser_claude import sdk_client
|
||||
from portfolio_optimiser_claude.contracts import ModelMapContract
|
||||
from portfolio_optimiser_claude.sdk_client import SdkModelClient, build_call_options
|
||||
from portfolio_optimiser_claude.sdk_client import SdkModelClient, _total_tokens, build_call_options
|
||||
|
||||
|
||||
class TestBuildCallOptions:
|
||||
|
|
@ -32,8 +33,11 @@ class TestBuildCallOptions:
|
|||
assert options.setting_sources == []
|
||||
|
||||
def test_the_system_prompt_is_empty(self) -> None:
|
||||
# None serializes to --system-prompt "" (verified against 0.2.110):
|
||||
# no Claude Code preset, no appended operator instructions.
|
||||
# Pins the OPTION value: None, not the Claude Code preset. That None
|
||||
# reaches the spawned CLI as --system-prompt "" was verified by
|
||||
# READING subprocess_cli.py (0.2.110–0.2.120) — this test does NOT
|
||||
# bind that transport serialization; doing so would couple the suite
|
||||
# to SDK-private API (the F11 fragility this repo retired).
|
||||
options = build_call_options("model-x", max_budget_usd=0.25)
|
||||
assert options.system_prompt is None
|
||||
|
||||
|
|
@ -75,3 +79,148 @@ class TestCompleteThreadsIsolatedOptions:
|
|||
assert captured["options"].model == "model-default"
|
||||
# No usage surfaced by the fake → the reply fails CLOSED (§8).
|
||||
assert reply.usage_tokens is None
|
||||
|
||||
|
||||
def _stream_of(*messages: Any) -> Any:
|
||||
"""A fake ``query`` yielding a scripted stream of REAL SDK message objects."""
|
||||
|
||||
def fake_query(*, prompt: str, options: Any) -> AsyncIterator[Any]:
|
||||
async def _stream() -> AsyncIterator[Any]:
|
||||
for message in messages:
|
||||
yield message
|
||||
|
||||
return _stream()
|
||||
|
||||
return fake_query
|
||||
|
||||
|
||||
def _assistant(*blocks: Any, model: str = "model-real", error: Any = None) -> AssistantMessage:
|
||||
return AssistantMessage(content=list(blocks), model=model, error=error)
|
||||
|
||||
|
||||
def _result(
|
||||
usage: dict[str, Any] | None = None,
|
||||
total_cost_usd: float | None = None,
|
||||
is_error: bool = False,
|
||||
subtype: str = "success",
|
||||
errors: list[str] | None = None,
|
||||
) -> ResultMessage:
|
||||
return ResultMessage(
|
||||
subtype=subtype,
|
||||
duration_ms=1,
|
||||
duration_api_ms=1,
|
||||
is_error=is_error,
|
||||
num_turns=1,
|
||||
session_id="s",
|
||||
usage=usage,
|
||||
total_cost_usd=total_cost_usd,
|
||||
errors=errors,
|
||||
)
|
||||
|
||||
|
||||
def _client() -> SdkModelClient:
|
||||
return SdkModelClient(ModelMapContract(profiles={"anthropic": {"default": "model-default"}}))
|
||||
|
||||
|
||||
_FULL_USAGE = {
|
||||
"input_tokens": 10,
|
||||
"output_tokens": 5,
|
||||
"cache_creation_input_tokens": 3,
|
||||
"cache_read_input_tokens": 2,
|
||||
}
|
||||
|
||||
|
||||
class TestCompleteAsyncStreamBinding:
|
||||
"""C2.5 (R-4/R-5): the read loop is BOUND offline with real SDK message types.
|
||||
|
||||
Before C2.5 nothing in the suite executed sdk_client's aggregation,
|
||||
error, usage or cost branches — the fake stream (real ``AssistantMessage``
|
||||
/ ``ResultMessage`` / ``TextBlock`` objects, so constructor drift also
|
||||
goes red) binds every branch without a key or the network.
|
||||
"""
|
||||
|
||||
def test_text_aggregates_and_non_text_blocks_are_ignored(
|
||||
self, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
monkeypatch.setattr(
|
||||
sdk_client,
|
||||
"query",
|
||||
_stream_of(
|
||||
_assistant(TextBlock("{"), ThinkingBlock(thinking="hmm", signature="sig")),
|
||||
_assistant(TextBlock("}")),
|
||||
_result(usage=_FULL_USAGE, total_cost_usd=0.01),
|
||||
),
|
||||
)
|
||||
client = _client()
|
||||
reply = client.complete("p", role="proposer")
|
||||
assert reply.text == "{}"
|
||||
assert reply.model == "model-real"
|
||||
assert client.last_model == "model-real"
|
||||
|
||||
def test_usage_tokens_sum_the_four_provider_fields(
|
||||
self, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
monkeypatch.setattr(
|
||||
sdk_client,
|
||||
"query",
|
||||
_stream_of(_assistant(TextBlock("ok")), _result(usage=_FULL_USAGE)),
|
||||
)
|
||||
assert _client().complete("p", role="proposer").usage_tokens == 20
|
||||
|
||||
def test_cost_accumulates_across_calls(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
client = _client()
|
||||
for cost in (0.01, 0.02):
|
||||
monkeypatch.setattr(
|
||||
sdk_client,
|
||||
"query",
|
||||
_stream_of(_assistant(TextBlock("ok")), _result(total_cost_usd=cost)),
|
||||
)
|
||||
client.complete("p", role="proposer")
|
||||
assert client.total_cost_usd == pytest.approx(0.03)
|
||||
|
||||
def test_an_assistant_error_fails_the_call(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(
|
||||
sdk_client, "query", _stream_of(_assistant(TextBlock("x"), error="rate_limit"))
|
||||
)
|
||||
with pytest.raises(RuntimeError, match="rate_limit"):
|
||||
_client().complete("p", role="proposer")
|
||||
|
||||
def test_a_result_error_fails_the_call_naming_subtype_and_errors(
|
||||
self, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
monkeypatch.setattr(
|
||||
sdk_client,
|
||||
"query",
|
||||
_stream_of(
|
||||
_assistant(TextBlock("x")),
|
||||
_result(is_error=True, subtype="error_during_execution", errors=["boom"]),
|
||||
),
|
||||
)
|
||||
with pytest.raises(RuntimeError, match="error_during_execution.*boom"):
|
||||
_client().complete("p", role="proposer")
|
||||
|
||||
|
||||
class TestTotalTokensFailsClosed:
|
||||
"""§8: the meter is never fed an invented count — no usage stays ``None``."""
|
||||
|
||||
def test_no_usage_dict_is_none(self) -> None:
|
||||
assert _total_tokens(None) is None
|
||||
|
||||
def test_an_empty_usage_dict_is_none(self) -> None:
|
||||
assert _total_tokens({}) is None
|
||||
|
||||
def test_non_int_fields_are_ignored_not_coerced(self) -> None:
|
||||
assert _total_tokens({"input_tokens": "10"}) is None
|
||||
assert _total_tokens({"input_tokens": 10, "output_tokens": "x"}) == 10
|
||||
|
||||
|
||||
class TestBudgetGuard:
|
||||
"""§8: a non-positive per-call USD cap is refused at construction."""
|
||||
|
||||
@pytest.mark.parametrize("cap", [0.0, -0.5])
|
||||
def test_non_positive_caps_are_rejected(self, cap: float) -> None:
|
||||
with pytest.raises(ValueError):
|
||||
SdkModelClient(
|
||||
ModelMapContract(profiles={"anthropic": {"default": "m"}}),
|
||||
max_budget_usd_per_call=cap,
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue