331 lines
14 KiB
Python
331 lines
14 KiB
Python
"""Offline simulation of the full agentic loop — the end-to-end method proof (replaces målbilde
|
|
§11.8's real-model run).
|
|
|
|
**Operator decision (pragmatic, cost-driven):** MAF is NOT run against a real model — neither
|
|
Azure/Foundry nor local Ollama — because paying for API runs across both repos (MAF + the
|
|
Claude-SDK sibling) is too costly privately. This module is the primary proof instead: it drives
|
|
``run_project`` with a **scripted** synthetic chat client (no network, no model) and demonstrates
|
|
that the loop's dataflow closes end to end across two runs separated by a promotion:
|
|
|
|
context -> hypothesis -> maker/checker debate -> deterministic validator -> persona verdict
|
|
-> PROMOTION into the OKF wiki -> the next run's hypothesis is informed by it.
|
|
|
|
**What this proves:** the plumbing, the deterministic spine, and that the learning loop closes —
|
|
a verdict approved in Run A reaches Run B's hypothesis prompt purely through the file-backed wiki.
|
|
**What it does NOT prove (honesty, målbilde §1):** that a live LLM would *produce* the proposal or
|
|
the verdict unprompted — those are scripted stand-ins for the swarm and the expert persona. The
|
|
genuine model-behaviour comparison lives on the Claude-SDK side (a minimal API run). The scripted
|
|
client is MAF-side scaffolding; it is NOT part of the framework-neutral ``shared/`` core.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import shutil
|
|
from collections.abc import Awaitable, Callable, Mapping, Sequence
|
|
from dataclasses import dataclass
|
|
from pathlib import Path
|
|
from typing import Any, cast
|
|
|
|
from agent_framework import (
|
|
BaseChatClient,
|
|
ChatResponse,
|
|
ChatResponseUpdate,
|
|
Message,
|
|
ResponseStream,
|
|
UsageDetails,
|
|
)
|
|
from agent_framework_openai import OpenAIChatCompletionClient
|
|
|
|
from portfolio_optimiser.persona import load_persona_example
|
|
from portfolio_optimiser.run import RunResult, run_project
|
|
from portfolio_optimiser.shared_root import shared_root
|
|
from portfolio_optimiser.validator import ValidatedProposal
|
|
from portfolio_optimiser.verdicts import VerdictStore, promote_verdict, seed_store_from_bundle
|
|
|
|
_PROJECT_ID = "BYGG-KONTOR-NORD"
|
|
|
|
|
|
def _default_bundle_dir() -> Path:
|
|
"""The demo bundle under the shared core, resolved at CALL time via ``shared_root()`` (env
|
|
``PORTFOLIO_SHARED_ROOT`` re-points it — the S4 extraction seam)."""
|
|
return shared_root() / "examples" / "bygg-energi-mikro"
|
|
|
|
|
|
# A VALID SavingsProposal for BYGG-KONTOR-NORD: total = 300000 x 1.0, P90 = 0.30 x 300000 = 90000,
|
|
# claimed 30000 <= 90000 -> validates on the first attempt (no `assumptions` -> degenerate MC).
|
|
_VALID_PROPOSAL = (
|
|
'{"measure":"LED-retrofit av kontorbelysning","affected_items":'
|
|
'[{"code":"ENERGI-TOTAL-EL","quantity":300000,"unit_cost":1.0}],"claimed_saving_nok":30000}'
|
|
)
|
|
# The checker's debate turn ends with the gate marker the run parses (run._checker_verdict).
|
|
_CHECKER_APPROVE = "Tallene er innenfor feasibelt område og resonnementet holder. VERDICT: APPROVE"
|
|
|
|
# The persona's verdict is sourced from the shared expert-reviewer skill (``load_persona_example``),
|
|
# NOT inlined here — that de-stubs the persona and makes the shared artifact genuinely consumed. Its
|
|
# marker (a realization rate ABSENT from the bundle — the seed is 0.82) is the payload we trace from
|
|
# Run A's persona judgement, through promotion, into Run B's hypothesis prompt.
|
|
|
|
|
|
class ScriptedChatClient(OpenAIChatCompletionClient):
|
|
"""The ONE canonical network-free scripted chat client (S2.5 consolidation): a single
|
|
``_inner_get_response`` body shared by the simulation's fixed-reply client AND conftest's three
|
|
test doubles (which subclass it). Parametrized by a ``reply_selector`` over
|
|
``(prompt_blob, role)`` plus an optional ``sink`` recording every prompt — the four
|
|
previously-divergent ``_inner_get_response`` bodies collapse to this one.
|
|
|
|
Subclasses the LAYERED ``OpenAIChatCompletionClient`` (not the minimal ``BaseChatClient``) so the
|
|
always-attached ``BudgetMiddleware`` is not silently no-op'd (verified). Construction is offline
|
|
(loopback ``base_url`` + dummy key); ``_inner_get_response`` intercepts before any HTTP.
|
|
|
|
Back-compat constructors are preserved (divergent PUBLIC surfaces the external test call-sites
|
|
depend on): ``ScriptedChatClient(reply, sink)`` (POSITIONAL — used by ``scripted_factory``) is
|
|
sugar for a constant selector; the ``call_count`` attribute + ``model``/OTEL ``"synthetic"`` are
|
|
always present; subclasses pass ``reply_selector=`` / ``default_reply=`` for scripted-list,
|
|
prompt-scan, or record-only behaviour."""
|
|
|
|
OTEL_PROVIDER_NAME = "synthetic"
|
|
|
|
def __init__(
|
|
self,
|
|
reply: str | None = None,
|
|
sink: list[str] | None = None,
|
|
*,
|
|
reply_selector: Callable[[str, str], str] | None = None,
|
|
role: str = "",
|
|
default_reply: str = "ok",
|
|
tokens_per_reply: int = 8,
|
|
) -> None:
|
|
super().__init__(model="synthetic", api_key="synthetic", base_url="http://127.0.0.1:9/v1")
|
|
self._sink = sink
|
|
self._role = role
|
|
self._default = default_reply
|
|
# The reply-selector over (prompt_blob, role). A positional ``reply`` is sugar for a constant
|
|
# selector (scripted_factory back-compat); with neither, the constant is ``default_reply``.
|
|
if reply_selector is not None:
|
|
self._select: Callable[[str, str], str] = reply_selector
|
|
elif reply is not None:
|
|
self._select = lambda _prompt, _role: reply
|
|
else:
|
|
self._select = lambda _prompt, _role: self._default
|
|
self._tokens = tokens_per_reply
|
|
self.call_count = 0
|
|
|
|
def _inner_get_response(
|
|
self,
|
|
*,
|
|
messages: Sequence[Message],
|
|
options: Mapping[str, Any],
|
|
stream: bool = False,
|
|
**kwargs: Any,
|
|
) -> Awaitable[ChatResponse] | ResponseStream[ChatResponseUpdate, ChatResponse]:
|
|
blob = " ".join(getattr(m, "text", "") or "" for m in messages)
|
|
if self._sink is not None:
|
|
self._sink.append(blob)
|
|
self.call_count += 1
|
|
reply = self._select(blob, self._role)
|
|
usage = UsageDetails(total_token_count=self._tokens)
|
|
if stream:
|
|
|
|
async def _agen() -> Any:
|
|
# The framework accepts a {"type": "text", ...} dict here (its types under-specify it).
|
|
yield ChatResponseUpdate(
|
|
role="assistant",
|
|
contents=[{"type": "text", "text": reply}], # type: ignore[list-item]
|
|
)
|
|
|
|
return self._build_response_stream(_agen())
|
|
|
|
async def _coro() -> ChatResponse:
|
|
return ChatResponse(
|
|
messages=[Message(role="assistant", contents=[reply])],
|
|
response_id="synthetic",
|
|
usage_details=usage,
|
|
)
|
|
|
|
return _coro()
|
|
|
|
|
|
def scripted_factory(replies: dict[str, str], sink: list[str]) -> Callable[[str], BaseChatClient]:
|
|
"""A role-keyed client factory: ``factory("proposer")`` and ``factory("checker")`` each return a
|
|
fresh ``ScriptedChatClient`` with that role's reply, all sharing ONE ``sink``. MAF stamps the
|
|
proposer/checker identity from the agent name, so role-keyed stateless replies suffice (no
|
|
per-turn counter); the shared ``sink`` spans the debate turns and the generation call."""
|
|
|
|
def factory(role: str) -> BaseChatClient:
|
|
return ScriptedChatClient(replies[role], sink, role=role)
|
|
|
|
return factory
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class LearningSimulationResult:
|
|
"""The trace of one two-run learning simulation. ``marker_in_run_b_prompt`` true while
|
|
``marker_in_run_a_prompt`` false is the closed loop: the persona knowledge approved in Run A
|
|
reached Run B's hypothesis only via promotion into the wiki."""
|
|
|
|
run_a: RunResult
|
|
run_b: RunResult
|
|
promoted_path: Path
|
|
marker: str
|
|
marker_in_run_a_prompt: bool
|
|
marker_in_run_b_prompt: bool
|
|
run_a_generation_prompts: list[str]
|
|
run_b_generation_prompts: list[str]
|
|
|
|
|
|
def _generation_prompts(sink: list[str]) -> list[str]:
|
|
"""The generation-call prompts (``generate._build_messages`` embeds 'SavingsProposal'), isolated
|
|
from the debate-round prompts also captured in the shared sink."""
|
|
return [p for p in sink if "SavingsProposal" in p]
|
|
|
|
|
|
async def simulate_learning_loop(
|
|
bundle_dir: str,
|
|
work_dir: str,
|
|
*,
|
|
persona_rationale: str | None = None,
|
|
marker: str | None = None,
|
|
timestamp: str = "2026-06-30",
|
|
max_rounds: int = 3,
|
|
) -> LearningSimulationResult:
|
|
"""Run the loop twice on a throwaway COPY of the bundle (the shared fixture is never mutated),
|
|
with a promotion in between, and trace whether the persona's approved knowledge crosses runs.
|
|
|
|
The persona's verdict (decision + rationale + traced ``marker``) defaults to the shared
|
|
expert-reviewer skill's canonical example (``load_persona_example``), read at CALL time — so the
|
|
simulation is genuinely artifact-driven, not inlined. Callers may override ``marker`` /
|
|
``persona_rationale`` for a control.
|
|
|
|
Run A: a fresh (empty) wiki -> an uninformed hypothesis; the persona approves with NEW realization
|
|
knowledge (``marker`` in ``persona_rationale``). ``promote_verdict`` lifts that verdict into the
|
|
wiki; ``seed_store_from_bundle`` re-reads the wiki; Run B's Step-1 ExpeL fold then carries the
|
|
marker into its hypothesis prompt. The two runs use SEPARATE sinks so each prompt set is
|
|
inspected independently."""
|
|
example = load_persona_example()
|
|
if marker is None:
|
|
marker = example.marker
|
|
if persona_rationale is None:
|
|
persona_rationale = example.rationale
|
|
if marker not in persona_rationale:
|
|
raise ValueError("marker must be a substring of persona_rationale (the carried payload)")
|
|
|
|
copy = Path(work_dir) / "bundle"
|
|
shutil.copytree(bundle_dir, copy)
|
|
copy_s = str(copy)
|
|
replies = {"proposer": _VALID_PROPOSAL, "checker": _CHECKER_APPROVE}
|
|
verdict_input = {"decision": example.decision, "rationale": persona_rationale}
|
|
|
|
# Run A — empty wiki isolates the persona's NEW knowledge.
|
|
sink_a: list[str] = []
|
|
run_a = cast(
|
|
RunResult, # the sim only drives full runs; never dry-run (S4.2 widened run_project)
|
|
await run_project(
|
|
_PROJECT_ID,
|
|
"local",
|
|
docs_dir=copy_s,
|
|
bundle_dir=copy_s,
|
|
verdict_input=verdict_input,
|
|
store=VerdictStore(verdicts=[]),
|
|
client_factory=scripted_factory(replies, sink_a),
|
|
max_rounds=max_rounds,
|
|
),
|
|
)
|
|
|
|
# Gate-promote the persona verdict from the raw output layer into the OKF wiki (Steg 8).
|
|
promoted_path = promote_verdict(
|
|
copy_s,
|
|
run_a.verdict,
|
|
approver="ekspert-persona (sim)",
|
|
experiment="sim-run-A",
|
|
timestamp=timestamp,
|
|
)
|
|
|
|
# Re-seed the wiki: the promoted verdict is now navigable and folds into the next run.
|
|
store_b = seed_store_from_bundle(copy_s)
|
|
|
|
# Run B — a separate, later run reads the updated wiki.
|
|
sink_b: list[str] = []
|
|
run_b = cast(
|
|
RunResult, # the sim only drives full runs; never dry-run (S4.2 widened run_project)
|
|
await run_project(
|
|
_PROJECT_ID,
|
|
"local",
|
|
docs_dir=copy_s,
|
|
bundle_dir=copy_s,
|
|
verdict_input=verdict_input,
|
|
store=store_b,
|
|
client_factory=scripted_factory(replies, sink_b),
|
|
max_rounds=max_rounds,
|
|
),
|
|
)
|
|
|
|
gen_a = _generation_prompts(sink_a)
|
|
gen_b = _generation_prompts(sink_b)
|
|
return LearningSimulationResult(
|
|
run_a=run_a,
|
|
run_b=run_b,
|
|
promoted_path=promoted_path,
|
|
marker=marker,
|
|
marker_in_run_a_prompt=any(marker in p for p in gen_a),
|
|
marker_in_run_b_prompt=any(marker in p for p in gen_b),
|
|
run_a_generation_prompts=gen_a,
|
|
run_b_generation_prompts=gen_b,
|
|
)
|
|
|
|
|
|
def _outcome_line(result: RunResult) -> str:
|
|
o = result.outcome
|
|
if isinstance(o, ValidatedProposal):
|
|
return (
|
|
f"VALIDATED (claimed {o.proposal.claimed_saving_nok:.0f} <= P90 {o.p90:.0f} NOK; "
|
|
f"measure: {o.proposal.measure})"
|
|
)
|
|
return f"REJECTED ({o.reason})"
|
|
|
|
|
|
def main(argv: list[str] | None = None) -> int: # pragma: no cover - console trace
|
|
"""Run the simulation against the energi bundle in a throwaway temp dir and print an honest,
|
|
readable trace. Invoke: ``uv run python -m portfolio_optimiser.simulation``."""
|
|
import asyncio
|
|
import tempfile
|
|
|
|
work = tempfile.mkdtemp(prefix="po-sim-")
|
|
result = asyncio.run(simulate_learning_loop(str(_default_bundle_dir()), work))
|
|
|
|
print("=" * 78)
|
|
print("OFFLINE SIMULATION — scripted agent replies, NO real model.")
|
|
print("Proves the loop's dataflow + deterministic spine + that the learning loop closes.")
|
|
print("Does NOT prove a live LLM would produce these — proposal/verdict are scripted.")
|
|
print("=" * 78)
|
|
|
|
print("\nRUN A (fresh wiki — no prior verdicts)")
|
|
print(f" validator : {_outcome_line(result.run_a)}")
|
|
print(f" checker : VERDICT={result.run_a.checker_verdict.upper()}")
|
|
print(f" persona : {result.run_a.verdict.decision} -> {result.run_a.verdict.rationale}")
|
|
print(
|
|
f" prompt has marker '{result.marker}': {result.marker_in_run_a_prompt} (expected False)"
|
|
)
|
|
|
|
print("\nPROMOTE (gated wiki-promotion, Steg 8)")
|
|
print(f" wrote : {result.promoted_path.name} (linked into index.md, neutral label)")
|
|
|
|
print("\nRUN B (re-seeded wiki — reads the promoted verdict)")
|
|
print(f" validator : {_outcome_line(result.run_b)}")
|
|
print(
|
|
f" prompt has marker '{result.marker}': {result.marker_in_run_b_prompt} (expected True)"
|
|
)
|
|
|
|
closed = result.marker_in_run_b_prompt and not result.marker_in_run_a_prompt
|
|
print("\n" + "-" * 78)
|
|
if closed:
|
|
print("LEARNING LOOP CLOSED: the persona knowledge approved in Run A reached Run B's")
|
|
print("hypothesis purely via the file-backed OKF wiki (promote -> re-seed -> ExpeL fold).")
|
|
else:
|
|
print("LEARNING LOOP NOT CLOSED — the marker did not cross runs as expected.")
|
|
print("-" * 78)
|
|
print(f"\n(working copy: {work})")
|
|
return 0 if closed else 1
|
|
|
|
|
|
if __name__ == "__main__": # pragma: no cover - console entry
|
|
raise SystemExit(main())
|