Transkript-analyse av den stoppede live-kjøringen (10 kall, 162 250 tokens, $0.331506): konfig-lekkasjen (setting_sources=None laster ALLE filsystem- settings) injiserte operatørens Claude-konfig i hvert kall — ~10-15k uncachede tokens, en påtvunget bekreftelses-preamble som gjorde ren-JSON-svar umulige, og en checker kapret av lekkede instrukser (debatt konvergerte aldri). - persist_stop_artifacts: stopp-event verbatim + usage/kost persisteres ALLTID ved BudgetExceeded (delt usage-shape med fullført-run-stien) - build_call_options: setting_sources=[] (SDK isolation mode, verifisert mot installert 0.2.110-kilde), system_prompt=None → tom system-prompt; detach- bevis via monkeypatchet query - _generation_prompt: krever ONLY the raw JSON object (fence-innpakning ga 4 fullpris parse-retries) 187/187 uten nøkkel · ruff + mypy --strict rene · tre detach-bevis RØDE → grønn Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01QdSfQdND84oeq2mbjueLTS
152 lines
5.5 KiB
Python
152 lines
5.5 KiB
Python
"""The S10 output layer: §9 citations + deterministic run-artifact persistence.
|
|
|
|
``build_citations`` cites the navigated read-context (never ``type: verdict``
|
|
files — the verdict layer must not leak, §3 Step 1) with EXACT character spans
|
|
into the source files; a context that yields no citable content fails fast (§9).
|
|
``persist_run_artifacts`` writes the captured run for S11: the decisions are
|
|
mirrored VERBATIM from the ``RunResult`` (§9 non-conflation — never recomputed
|
|
from a checker-overridden outcome), and the bytes are deterministic (sorted
|
|
keys, 2-space indent — the house JSON convention).
|
|
|
|
Pure file layer by design — imports no agent toolkit; the SDK client stays
|
|
run-path-only (§11: the suite runs without a key).
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from portfolio_optimiser_claude.budget import BudgetExceeded
|
|
from portfolio_optimiser_claude.contracts import TerminationContract
|
|
from portfolio_optimiser_claude.loop import RunResult
|
|
from portfolio_optimiser_claude.okf import ConceptFile
|
|
from portfolio_optimiser_claude.provenance import Citation, Provenance
|
|
from portfolio_optimiser_claude.validator import Rejection
|
|
|
|
_VERDICT_TYPE = "verdict"
|
|
|
|
|
|
def build_citations(concepts: list[ConceptFile]) -> list[Citation]:
|
|
"""One citation per citable concept: file + exact char span + snippet (§9).
|
|
|
|
The snippet is the concept body's first non-empty line; the span locates it
|
|
verbatim in the SOURCE file (frontmatter included), so every citation is
|
|
independently verifiable. Verdict files and empty bodies are skipped; a
|
|
context with no citable content raises (fail fast, §9).
|
|
"""
|
|
citations: list[Citation] = []
|
|
for concept in concepts:
|
|
if concept.type == _VERDICT_TYPE or not concept.body:
|
|
continue
|
|
snippet = next((line for line in concept.body.splitlines() if line.strip()), "")
|
|
if not snippet:
|
|
continue
|
|
source = concept.path.read_text(encoding="utf-8")
|
|
start = source.find(snippet)
|
|
if start < 0:
|
|
continue
|
|
citations.append(
|
|
Citation(
|
|
file=concept.path.name,
|
|
span=f"chars {start}-{start + len(snippet)}",
|
|
snippet=snippet,
|
|
)
|
|
)
|
|
if not citations:
|
|
raise ValueError("context yields no citable content — a run must fail fast (§9)")
|
|
return citations
|
|
|
|
|
|
def _dump_json(path: Path, payload: dict[str, Any]) -> None:
|
|
path.write_text(json.dumps(payload, sort_keys=True, indent=2) + "\n", encoding="utf-8")
|
|
|
|
|
|
def _usage_payload(
|
|
termination: TerminationContract,
|
|
tokens_used: int,
|
|
rounds_used: int,
|
|
cost_usd: float | None,
|
|
) -> dict[str, Any]:
|
|
# ONE usage shape for both outcomes (completed run and budget stop) —
|
|
# S11 must never branch on which path wrote the artifact.
|
|
return {
|
|
"tokens_used": tokens_used,
|
|
"rounds_used": rounds_used,
|
|
"max_tokens": termination.max_tokens,
|
|
"max_rounds": termination.max_rounds,
|
|
"cost_usd": cost_usd,
|
|
}
|
|
|
|
|
|
def persist_run_artifacts(
|
|
out_dir: Path,
|
|
*,
|
|
run: RunResult,
|
|
provenance: Provenance,
|
|
termination: TerminationContract,
|
|
tokens_used: int,
|
|
rounds_used: int,
|
|
cost_usd: float | None,
|
|
) -> dict[str, Path]:
|
|
"""Persist the run for S11: proposal, result, provenance stamp, usage-vs-caps.
|
|
|
|
``validator_decision`` and ``checker_decision`` are copied from the
|
|
``RunResult`` fields — the two falsifiers were recorded separately there
|
|
(§9) and recomputing either from the outcome would conflate them.
|
|
"""
|
|
out_dir.mkdir(parents=True, exist_ok=True)
|
|
outcome: dict[str, Any] = (
|
|
{"type": "rejected", "reason": run.outcome.reason}
|
|
if isinstance(run.outcome, Rejection)
|
|
else {"type": "validated", **run.outcome.model_dump()}
|
|
)
|
|
paths = {
|
|
"proposal": out_dir / "proposal.json",
|
|
"run_result": out_dir / "run_result.json",
|
|
"provenance": out_dir / "provenance.json",
|
|
"usage": out_dir / "usage.json",
|
|
}
|
|
_dump_json(paths["proposal"], run.proposal.model_dump())
|
|
_dump_json(
|
|
paths["run_result"],
|
|
{
|
|
"validator_decision": run.validator_decision,
|
|
"checker_decision": run.checker_decision,
|
|
"attempts": run.attempts,
|
|
"outcome": outcome,
|
|
},
|
|
)
|
|
_dump_json(paths["provenance"], provenance.model_dump())
|
|
_dump_json(paths["usage"], _usage_payload(termination, tokens_used, rounds_used, cost_usd))
|
|
return paths
|
|
|
|
|
|
def persist_stop_artifacts(
|
|
out_dir: Path,
|
|
*,
|
|
stop: BudgetExceeded,
|
|
termination: TerminationContract,
|
|
tokens_used: int,
|
|
rounds_used: int,
|
|
cost_usd: float | None,
|
|
) -> dict[str, Path]:
|
|
"""Persist a budget-stopped run: the stop event verbatim + usage-vs-caps.
|
|
|
|
A budget stop is a run outcome, not an absence of one (§8): the S10 live
|
|
run stopped structurally on the token cap and persisted NOTHING — real
|
|
spend without a record. The stop event mirrors the raised ``BudgetExceeded``
|
|
fields exactly; the usage artifact keeps the completed-run shape.
|
|
"""
|
|
out_dir.mkdir(parents=True, exist_ok=True)
|
|
paths = {
|
|
"stop": out_dir / "stop.json",
|
|
"usage": out_dir / "usage.json",
|
|
}
|
|
_dump_json(
|
|
paths["stop"],
|
|
{"kind": stop.kind, "limit": stop.limit, "observed": stop.observed},
|
|
)
|
|
_dump_json(paths["usage"], _usage_payload(termination, tokens_used, rounds_used, cost_usd))
|
|
return paths
|