feat(explore): U4 kallsted 1 - --explore former mandatet, og sporet overlever taket (ORDRE 20260823T204216Z) [skip-docs]

Kallsted (1) + artefaktet (4) av oerkt 57s fire.

--explore "<prompt>" + --explore-config FILE i run.py: utforskningen kjoerer FOER
pipelinen og mandatet den former gaar rett inn i run_project(mandate=...). Flagget
er opt-in, og hver ting det ikke kan aere NEKTES ved navn:

- --explore-config uten --explore (--embedder-config-presedensen, ordrett)
- --explore uten --explore-config: CLI-en oppfinner ALDRI grenser, fordi
  MagenticBuilders egen fallback er "ubegrenset"
- --explore + --mandate: TO KILDER TIL ETT MANDAT. Nektet, aldri slaatt sammen -
  explore() tar objective fra prompten og hardkoder allow_own_proposals=True, saa
  en sammenslaaing ville stille overskrevet tre felt operatoeren skrev selv.
  Nekten NAVNGIR biblioteksdoera (seed_approaches), fordi C.6 doer 1 er et ekte
  behov denne flaten ikke betjener
- --explore + --live-dry-run (motstrid), --explore uten --bundle-dir (leser
  ingenting), --portfolio --explore (partisjonen)
- enable_plan_review=true nektes HER, ikke i explore(): ExplorationError er en
  RuntimeError og ligger UTENFOR main()s (ValueError, FileNotFoundError,
  ValidationError)-tuppel, saa den ville forlatt som traceback i stedet for rc 1

ExplorationTrace er en KALLER-EID akkumulator (funn-1-sinken, ett lag opp):
explore() raiser BudgetExceeded paa rundetaket og tokentaket fyrer fra middleware
midt i loepet - paa begge stier finnes ingen ExplorationResult, og C.2 krever at
artefaktet er lesbart uansett hvilken vakt som fyrte. ExplorationResult.ledger_log
BYGGES FRA akkumulatoren, aldri ved siden av (kø-(p)).

{run_id}-exploration.json skrives fra en finally (write_parse_failures-presedensen)
med rundene, plan-reviewene og quick_validate-dommene - de siste bor bevisst ikke i
ExplorationResult. `completed` er et eget felt: en stop: null som betyr BAADE
"avsluttet normalt" og "vi fikk aldri vite" er stillheten cost_baseline_anchored
ble paakrevd for aa lukke.

Load-bearing MAALT (tests/test_explore_callsites_loadbearing.py, 15 tester), ti
mutasjoner alle roede mot HELE suiten + groenn kontroll 990/5: detach sink-appenden
(1 roed) - andre liste for rundene (4 roede) - detach --mandate-nekten (1) - detach
--explore-config-nekten (1) - skriv artefaktet kun naar kjoeringen fullfoerte (1) -
detach mandate= inn i run_project (1) - slipp enable_plan_review gjennom (1) -
detach --bundle-dir-kravet (1) - detach --live-dry-run-nekten (1) - fjern --explore
fra portefoelje-partisjonen (1).

EN MUTASJON FALSIFISERTE TESTEN FOERST (repoets vakuoes-gate-klasse, syvende gang):
portefoelje-testen asserterte kun at meldingen nevnte --explore, og var groenn UTEN
partisjonen - kjoeringen falt da gjennom til "--explore requires --bundle-dir", som
nevner --explore ogsaa. To nekter som deler en delstreng; testen navngir naa
--portfolio.

Golden-transkriptet byte-uendret (ea8c534773acdbe41ae68f2c55724d69aaf8be4f).
mypy + ruff rene.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01YRZhBJcxqTcqWyMW6hBttx
This commit is contained in:
Kjell Tore Guttormsen 2026-08-25 09:08:49 +02:00
commit 8ba824c96f
4 changed files with 897 additions and 21 deletions

View file

@ -25,7 +25,7 @@ from __future__ import annotations
import json import json
from collections.abc import Callable, Mapping, Sequence from collections.abc import Callable, Mapping, Sequence
from dataclasses import dataclass from dataclasses import dataclass, field
from pathlib import Path from pathlib import Path
from typing import Any, Final, Literal from typing import Any, Final, Literal
@ -103,6 +103,40 @@ class ExplorationContract(BaseModel):
return self return self
def exploration_notice(result: "ExplorationResult") -> str:
"""The ONE line an exploration says about itself before the pipeline takes over.
Always a line, never an omission unlike ``cost_baseline_notice`` and
``skipped_links_notice``, which qualify a run that happened anyway. This one is printed only on
a surface where ``--explore`` was asked for, so there is no run it could speak for silently.
``stop`` is named when there is one, because a mandate shaped by a loop that gave up is a
smaller mandate than the same loop finishing would have produced, and nothing else on stdout
would say so.
"""
ended = "concluded" if result.stop is None else f"stopped ({result.stop})"
return (
f"Exploration: {ended} after {len(result.ledger_log)} round(s); "
f"{len(result.mandate.approaches)} approach(es) to evaluate"
)
def load_exploration_contract(path: str | Path) -> ExplorationContract:
"""Fail-fast standalone loader for an exploration's bounds (mirrors ``mandate.load_mandate``).
The bounds are authoritative startup input, and the ONE thing that keeps a Magentic loop from
being unbounded so a missing or malformed file refuses the run rather than degrading to
defaults, which is the shape ``ExplorationContract`` refuses to have in the first place.
:raises FileNotFoundError: ``path`` does not point at an existing file.
:raises pydantic.ValidationError: the content is not JSON, or violates the contract.
"""
p = Path(path)
if not p.is_file():
raise FileNotFoundError(f"exploration config not found: {str(path)!r}")
return ExplorationContract.model_validate_json(p.read_text(encoding="utf-8"))
#: The manager's role name. Not a participant: the manager plans, picks the next speaker and #: The manager's role name. Not a participant: the manager plans, picks the next speaker and
#: keeps the progress ledger, and ``BudgetMiddleware`` rides on it like on everyone else (A1, #: keeps the progress ledger, and ``BudgetMiddleware`` rides on it like on everyone else (A1,
#: measured green — without that, the most talkative agent in the loop would be the one outside #: measured green — without that, the most talkative agent in the loop would be the one outside
@ -186,6 +220,89 @@ class PlanReview:
feedback: str = "" feedback: str = ""
@dataclass(frozen=True)
class QuickValidation:
"""One advisory (level 1) verdict, as the tool answered it.
``proposal_json`` is kept VERBATIM for the reason the parse-failure capture keeps the raw reply
verbatim: the operative question after a run is what the hypothesiser actually asked about, and
a re-serialised form would answer a different one.
"""
bundle_id: str
proposal_json: str
verdict: Mapping[str, Any]
@dataclass
class ExplorationTrace:
"""The CALLER-owned accumulator for everything one exploration produced along the way.
**Why a caller-owned sink and not a return value** the same measurement that shaped the
parse-failure capture (Fase 1b, funn 1), one layer up. ``explore()`` raises ``BudgetExceeded``
at its round cap, and a token cap fires from inside ``BudgetMiddleware`` mid-run; on both paths
no ``ExplorationResult`` is ever constructed. § C.2 requires the exploration artefact to be
readable "uansett hvilken vakt som fyrte", so the evidence has to live somewhere that survives
the ending which a return value, by definition, does not. Step 5's opposite rule ("a returned
value, never an out-parameter") governs a value that DOES reach the caller; copying it blindly
here would rebuild the very defect it was written against.
``ExplorationResult.ledger_log`` and ``.plan_reviews`` are built FROM these lists rather than
accumulated beside them: two containers holding one fact drift (-(p)), and a drifted pair
would let the returned result and the written artefact describe different runs.
"""
ledger: list[LedgerEntry] = field(default_factory=list)
plan_reviews: list[PlanReview] = field(default_factory=list)
quick_validations: list[QuickValidation] = field(default_factory=list)
def trace_payload(trace: ExplorationTrace, *, stop: str | None, completed: bool) -> dict[str, Any]:
"""The ONE rendering of a trace into plain data for ``outbox.write_exploration``.
Plain mappings only, so the RAW output layer stays MAF-free (the ``write_parse_failures``
precedent ``outbox.py`` may not import this module).
``completed`` is a required field rather than an inference from ``stop``. With no result there
is no stop, and a ``stop: null`` meaning BOTH "concluded normally" and "we never found out"
is exactly the silence ``ProvenanceStamp.cost_baseline_anchored`` was made required to close.
"""
return {
"completed": completed,
"stop": stop,
"rounds": [
{
"round_index": entry.round_index,
"is_request_satisfied": entry.is_request_satisfied,
"is_in_loop": entry.is_in_loop,
"is_progress_being_made": entry.is_progress_being_made,
"next_speaker": entry.next_speaker,
"instruction_or_question": entry.instruction_or_question,
"speaker_known": entry.speaker_known,
}
for entry in trace.ledger
],
"plan_reviews": [
{
"index": review.index,
"plan": review.plan,
"is_stalled": review.is_stalled,
"decision": review.decision,
"feedback": review.feedback,
}
for review in trace.plan_reviews
],
"quick_validations": [
{
"bundle_id": call.bundle_id,
"proposal_json": call.proposal_json,
"verdict": dict(call.verdict),
}
for call in trace.quick_validations
],
}
#: Why an exploration ended without a satisfied request. ``None`` means it concluded normally. #: Why an exploration ended without a satisfied request. ``None`` means it concluded normally.
#: Resource exhaustion is NOT here: tokens and rounds raise ``BudgetExceeded`` (the 429 channel), #: Resource exhaustion is NOT here: tokens and rounds raise ``BudgetExceeded`` (the 429 channel),
#: because "we ran out" and "we finished, unsatisfied" are the two things S3.4 split apart and a #: because "we ran out" and "we finished, unsatisfied" are the two things S3.4 split apart and a
@ -341,7 +458,9 @@ def navigator_tools(bundle_dirs: Sequence[str]) -> list[FunctionTool]:
return [list_bundles, read_bundle, read_file] return [list_bundles, read_bundle, read_file]
def quick_validate_tool(bundle_dirs: Sequence[str]) -> FunctionTool: def quick_validate_tool(
bundle_dirs: Sequence[str], *, sink: list[QuickValidation] | None = None
) -> FunctionTool:
"""The hypothesiser's in-loop deterministic check — level 1, and advisory by construction. """The hypothesiser's in-loop deterministic check — level 1, and advisory by construction.
It is the SAME ``validate_proposal`` against the SAME baseline the pipeline will use, so the It is the SAME ``validate_proposal`` against the SAME baseline the pipeline will use, so the
@ -356,6 +475,11 @@ def quick_validate_tool(bundle_dirs: Sequence[str]) -> FunctionTool:
``ProvenanceStamp.cost_baseline_anchored`` is a required field: a verdict reached without the ``ProvenanceStamp.cost_baseline_anchored`` is a required field: a verdict reached without the
project's own cost lines is a weaker claim, and one that does not say so is the silence S4.0's project's own cost lines is a weaker claim, and one that does not say so is the silence S4.0's
visibility work closed. visibility work closed.
``sink`` is the caller's ``ExplorationTrace.quick_validations``. These verdicts are the one
thing ``ExplorationResult`` deliberately does not carry they are level-1 advisory, and their
home is the ``{run_id}-exploration.json`` artefact so without a sink they would leave no
trace of having been asked for at all.
""" """
index = _bundle_index(bundle_dirs) index = _bundle_index(bundle_dirs)
@ -370,25 +494,40 @@ def quick_validate_tool(bundle_dirs: Sequence[str]) -> FunctionTool:
def quick_validate(bundle_id: str, proposal_json: str) -> dict[str, Any]: def quick_validate(bundle_id: str, proposal_json: str) -> dict[str, Any]:
bundle_dir = _resolve_bundle(index, bundle_id) bundle_dir = _resolve_bundle(index, bundle_id)
baseline = okf.load_optional_cost_baseline(bundle_dir) baseline = okf.load_optional_cost_baseline(bundle_dir)
verdict: dict[str, Any]
try: try:
proposal = SavingsProposal.model_validate_json(proposal_json) proposal = SavingsProposal.model_validate_json(proposal_json)
except ValidationError as exc: except ValidationError as exc:
return {"decision": "unparseable", "reason": str(exc), "anchored": baseline is not None} verdict = {
outcome = validate_proposal(proposal, baseline=baseline) "decision": "unparseable",
if isinstance(outcome, Rejection): "reason": str(exc),
return {
"decision": "rejected",
"reason": outcome.reason,
"anchored": baseline is not None, "anchored": baseline is not None,
} }
return { else:
"decision": "validated", outcome = validate_proposal(proposal, baseline=baseline)
"reason": "", if isinstance(outcome, Rejection):
"anchored": baseline is not None, verdict = {
"p10": outcome.p10, "decision": "rejected",
"p50": outcome.p50, "reason": outcome.reason,
"p90": outcome.p90, "anchored": baseline is not None,
} }
else:
verdict = {
"decision": "validated",
"reason": "",
"anchored": baseline is not None,
"p10": outcome.p10,
"p50": outcome.p50,
"p90": outcome.p90,
}
# Recorded AFTER the verdict is decided and on EVERY branch — an unparseable candidate is
# as much a thing the hypothesiser asked about as a validated one. A refused bundle id
# raises above and is deliberately not recorded: nothing was validated.
if sink is not None:
sink.append(
QuickValidation(bundle_id=bundle_id, proposal_json=proposal_json, verdict=verdict)
)
return verdict
return quick_validate return quick_validate
@ -404,6 +543,7 @@ def fresh_exploration_workflow(
contract: ExplorationContract, contract: ExplorationContract,
bundle_dirs: Sequence[str] = (), bundle_dirs: Sequence[str] = (),
middleware: Sequence[Any] | None = None, middleware: Sequence[Any] | None = None,
quick_validate_sink: list[QuickValidation] | None = None,
) -> Any: ) -> Any:
"""Build a FRESH Magentic workflow with fresh agents and fresh clients (mirrors """Build a FRESH Magentic workflow with fresh agents and fresh clients (mirrors
``workflow.fresh_workflow``). ``workflow.fresh_workflow``).
@ -426,7 +566,7 @@ def fresh_exploration_workflow(
token guarantee: agent-level ``ChatMiddleware`` does fire on the manager's own calls (measured token guarantee: agent-level ``ChatMiddleware`` does fire on the manager's own calls (measured
A1), and the manager is the most talkative participant in the loop. A1), and the manager is the most talkative participant in the loop.
""" """
hypothesiser_tools: list[Any] = [quick_validate_tool(bundle_dirs)] hypothesiser_tools: list[Any] = [quick_validate_tool(bundle_dirs, sink=quick_validate_sink)]
tools_by_role: dict[str, list[Any]] = { tools_by_role: dict[str, list[Any]] = {
NAVIGATOR_ROLE: list(navigator_tools(bundle_dirs)), NAVIGATOR_ROLE: list(navigator_tools(bundle_dirs)),
HYPOTHESISER_ROLE: hypothesiser_tools, HYPOTHESISER_ROLE: hypothesiser_tools,
@ -631,6 +771,7 @@ async def explore(
plan_reviewer: PlanReviewer | None = None, plan_reviewer: PlanReviewer | None = None,
meter: TokenMeter | None = None, meter: TokenMeter | None = None,
success_criteria: str = "", success_criteria: str = "",
trace: ExplorationTrace | None = None,
) -> ExplorationResult: ) -> ExplorationResult:
"""Explore the knowledge bases and return the ``Mandate`` the pipeline should evaluate. """Explore the knowledge bases and return the ``Mandate`` the pipeline should evaluate.
@ -656,8 +797,13 @@ async def explore(
``bundle_dir``, and shipping the field before its consumer would be a shape guessed instead of ``bundle_dir``, and shipping the field before its consumer would be a shape guessed instead of
measured. (2) The ``quick_validate`` verdicts the hypothesiser saw are not in measured. (2) The ``quick_validate`` verdicts the hypothesiser saw are not in
``ExplorationResult``: they are level-1 advisory, and their home is the ``ExplorationResult``: they are level-1 advisory, and their home is the
``{run_id}-exploration.json`` artefact the CLI wiring writes. (3) The exploration roles resolve ``{run_id}-exploration.json`` artefact the CLI wiring writes pass an ``ExplorationTrace`` to
through ``resolve_model``'s ``default`` fallback unless an operator maps them explicitly. collect them. (3) The exploration roles resolve through ``resolve_model``'s ``default``
fallback unless an operator maps them explicitly.
``trace`` is the caller's accumulator and is the ONLY way to see what a run that RAISED
produced: both budget channels destroy the ``ExplorationResult`` before it exists. When it is
omitted a private one is used, so the returned result is unchanged for every existing caller.
""" """
if contract.enable_plan_review and plan_reviewer is None: if contract.enable_plan_review and plan_reviewer is None:
raise ExplorationError( raise ExplorationError(
@ -680,16 +826,22 @@ async def explore(
client_factory = _default_factory(profile) client_factory = _default_factory(profile)
if trace is None:
trace = ExplorationTrace()
workflow = fresh_exploration_workflow( workflow = fresh_exploration_workflow(
client_factory, client_factory,
contract=contract, contract=contract,
bundle_dirs=bundle_dirs, bundle_dirs=bundle_dirs,
middleware=[BudgetMiddleware(meter)], middleware=[BudgetMiddleware(meter)],
quick_validate_sink=trace.quick_validations,
) )
ledger_log: list[LedgerEntry] = [] # ONE accumulator per fact, held by the caller (see ``ExplorationTrace``). The local names are
# aliases, never copies — a second list here is the kø-(p) drift this shape exists to prevent.
ledger_log = trace.ledger
plan_reviews = trace.plan_reviews
hypothesis_texts: list[str] = [] hypothesis_texts: list[str] = []
plan_reviews: list[PlanReview] = []
seen: set[int] = set() seen: set[int] = set()
replans = 0 replans = 0
stop: ExplorationStop | None = None stop: ExplorationStop | None = None

View file

@ -165,6 +165,30 @@ def write_parse_failures(
return path return path
def write_exploration(
outbox_dir: str,
run_id: str,
*,
payload: Mapping[str, Any],
) -> Path:
"""Write ``{run_id}-exploration.json`` — what the U4 exploration did before the pipeline ran
(§ C.2) and return its path.
Takes an already-rendered plain mapping (``explore.trace_payload``) for the reason
``write_parse_failures`` takes plain mappings: ``explore`` imports ``agent_framework``, and
importing it here would drag MAF into the RAW output layer. The ONE renderer lives beside the
dataclasses it renders; this writer only decides bytes and a filename.
Byte-deterministic like its neighbours (the caller supplies ``run_id``; no wall-clock), and
written even when the exploration RAISED the caller writes it from a ``finally``, because a
capped exploration is precisely the one whose per-round ledger a reader needs."""
directory = Path(outbox_dir)
directory.mkdir(parents=True, exist_ok=True)
path = directory / f"{run_id}-exploration.json"
path.write_text(_dump({"run_id": run_id, **dict(payload)}), encoding="utf-8")
return path
def write_run_config( def write_run_config(
config_dir: str, config_dir: str,
run_id: str, run_id: str,

View file

@ -54,6 +54,15 @@ from portfolio_optimiser.datasource import (
retrieve_chunks, retrieve_chunks,
) )
from portfolio_optimiser.dimension import Dimension, admits, load_dimension from portfolio_optimiser.dimension import Dimension, admits, load_dimension
from portfolio_optimiser.explore import (
ExplorationContract,
ExplorationResult,
ExplorationTrace,
explore,
exploration_notice,
load_exploration_contract,
trace_payload,
)
from portfolio_optimiser.generate import ParseFailure, generate_via_llm from portfolio_optimiser.generate import ParseFailure, generate_via_llm
from portfolio_optimiser.ir import SavingsProposal from portfolio_optimiser.ir import SavingsProposal
from portfolio_optimiser.mandate import ( from portfolio_optimiser.mandate import (
@ -1411,6 +1420,27 @@ def main(argv: list[str] | None = None) -> int:
"afterwards, one row per approach. Valid in both modes; in portfolio mode it applies to " "afterwards, one row per approach. Valid in both modes; in portfolio mode it applies to "
"every project in the pass", "every project in the pass",
) )
parser.add_argument(
"--explore",
default=None,
metavar="PROMPT",
help="U4 opt-in: run a Magentic EXPLORATION over the knowledge base first and let it shape "
"the mandate this run then evaluates. The exploration chooses which base to open and which "
"directions are worth testing; every number it produces is still gated by the same "
"deterministic validator, and the exploration itself writes nothing. REQUIRES "
"--explore-config and --bundle-dir; refused together with --mandate (two sources of one "
"mandate)",
)
parser.add_argument(
"--explore-config",
default=None,
metavar="FILE",
help="the exploration's bounds (JSON, fail-fast, REQUIRES --explore): max_rounds, "
"max_tokens, max_stall_count, max_reset_count, max_plan_revisions, enable_plan_review. "
"Every field is required and none has a default — an omitted bound would fall back to "
"MAF's unbounded loop, not to something conservative. enable_plan_review must be false "
"here: the synchronous review has no reviewer on this surface",
)
parser.add_argument( parser.add_argument(
"--mcp-config", "--mcp-config",
default=None, default=None,
@ -1551,6 +1581,8 @@ def main(argv: list[str] | None = None) -> int:
"--semantic-retrieval": args.semantic_retrieval, "--semantic-retrieval": args.semantic_retrieval,
"--embedder-config": args.embedder_config is not None, "--embedder-config": args.embedder_config is not None,
"--scripted-replies": args.scripted_replies is not None, "--scripted-replies": args.scripted_replies is not None,
"--explore": args.explore is not None,
"--explore-config": args.explore_config is not None,
} }
if any(report_forbidden.values()): if any(report_forbidden.values()):
print( print(
@ -1590,6 +1622,12 @@ def main(argv: list[str] | None = None) -> int:
"--outbox-dir": args.outbox_dir, "--outbox-dir": args.outbox_dir,
"--run-id": args.run_id, "--run-id": args.run_id,
"--live-dry-run": args.live_dry_run, "--live-dry-run": args.live_dry_run,
# One exploration shapes ONE mandate against ONE knowledge base, and --bundle-dir (its
# only source of bases here) is already single-project-only. Refusing it by NAME beats
# letting it fall through to the --bundle-dir requirement below: an operator who wrote
# --portfolio --explore has to hear which of the two is wrong.
"--explore": args.explore,
"--explore-config": args.explore_config,
} }
offending = [name for name, value in single_only.items() if value] offending = [name for name, value in single_only.items() if value]
if offending: if offending:
@ -1673,6 +1711,84 @@ def main(argv: list[str] | None = None) -> int:
) )
return 1 return 1
# The exploration door (U4). Every refusal here is BY NAME and happens before anything runs,
# for the reason the whole block above exists: a flag that cannot take effect is refused, never
# silently ignored. Four of them, each closing a different way this could go quietly wrong.
#
# 1. --explore-config alone would be loaded and dropped (the --embedder-config case verbatim).
# 2. --explore alone has no bounds, and the CLI may not invent them: EVERY ExplorationContract
# field is required without a default precisely because MagenticBuilder's own fallback is
# "unbounded", which is the one shape shared/method-spec.md §8 forbids outright.
# 3. --explore + --mandate are TWO SOURCES OF ONE MANDATE. Refused rather than merged, and the
# decision is deliberate: ``explore()`` takes the objective from the prompt and hardcodes
# ``allow_own_proposals=True``, so composing them would silently overwrite three fields an
# operator wrote by hand — the silent merge this repo's flag contract forbids. § C.6 door 1
# (the expert's own hypotheses seeding the exploration) is a real need, and it is served by
# the library API; the refusal names it rather than only forbidding.
# 4. --explore + --live-dry-run contradict: the drill stops before the first model call and an
# exploration IS model calls (the --scripted-replies precedent, same words).
if args.explore_config is not None and args.explore is None:
print(
"run refused: --explore-config requires --explore (the bounds describe an exploration "
"that would never run, so the config would be loaded and then ignored)",
file=sys.stderr,
)
return 1
if args.explore is not None:
if args.explore_config is None:
print(
"run refused: --explore requires --explore-config (an exploration's bounds are "
"never defaulted — an omitted cap falls back to an unbounded loop, not to a "
"conservative one)",
file=sys.stderr,
)
return 1
if args.mandate is not None:
print(
"run refused: --explore and --mandate are two sources of one mandate. The "
"exploration SHAPES a mandate (objective from the prompt, own proposals allowed), "
"so merging would silently overwrite what you wrote. To seed an exploration with "
"an expert's own hypotheses, use the library door: "
"explore(..., seed_approaches=[Approach(...)])",
file=sys.stderr,
)
return 1
if args.live_dry_run:
print(
"run refused: --explore and --live-dry-run contradict each other (the drill stops "
"before the first model call; an exploration is model calls) — pick one",
file=sys.stderr,
)
return 1
if not args.bundle_dir:
print(
"run refused: --explore requires --bundle-dir (the exploration navigates knowledge "
"bases, and with none configured it would spend its budget reading nothing)",
file=sys.stderr,
)
return 1
exploration_contract: ExplorationContract | None = None
if args.explore_config is not None:
try:
exploration_contract = load_exploration_contract(args.explore_config)
except (FileNotFoundError, ValidationError, ValueError) as exc:
print(f"run refused: {exc}", file=sys.stderr)
return 1
if exploration_contract.enable_plan_review:
# Refused HERE rather than left to ``explore()``, which refuses it too: ExplorationError
# is a RuntimeError and therefore outside this CLI's (ValueError, FileNotFoundError,
# ValidationError) refusal tuple, so it would leave as a traceback instead of the rc 1
# line every other misconfiguration produces. The synchronous review (U13) needs a
# reviewer that blocks the loop, and this surface has none to offer.
print(
"run refused: --explore-config sets enable_plan_review, but this surface has no "
"reviewer to answer it (the run would stop at a review nobody can answer). The "
"synchronous door is the library API: explore(..., plan_reviewer=...)",
file=sys.stderr,
)
return 1
# The commission, loaded fail-fast BEFORE anything runs: a missing or malformed mandate is # The commission, loaded fail-fast BEFORE anything runs: a missing or malformed mandate is
# REFUSED rather than degraded to "no mandate", because the settlement would then describe work # REFUSED rather than degraded to "no mandate", because the settlement would then describe work
# nobody ordered. Placed with the other refusals and ABOVE the scripted banner, for the same # nobody ordered. Placed with the other refusals and ABOVE the scripted banner, for the same
@ -1732,6 +1848,52 @@ def main(argv: list[str] | None = None) -> int:
scripted_client_factory = scripted_factory(replies, []) scripted_client_factory = scripted_factory(replies, [])
print(_SCRIPTED_BANNER) print(_SCRIPTED_BANNER)
# U4: the exploration runs BEFORE the announcement, because what it produces IS the mandate the
# announcement describes. Its own model calls are therefore un-announced — stated plainly
# rather than papered over: the announcement's contract is that a COMMISSION is declared before
# the work it commissions, and until the exploration returns there is no commission to declare.
# ``exploration_notice`` is what covers the gap, printed the moment the loop is done.
#
# Every ExplorationError ``explore()`` can raise for a CONFIG reason is unreachable from here by
# construction: both plan-review preconditions are refused above, and the duplicate-base-id
# refusal needs two bases where this surface passes one. What can still escape — an unreadable
# marked hypothesis, an exhausted budget — is the RUN failing, not the caller erring, and leaves
# as it does for the debate today.
if args.explore is not None:
assert (
exploration_contract is not None
) # guarded above: --explore requires --explore-config
exploration_trace = ExplorationTrace()
exploration: ExplorationResult | None = None
try:
exploration = asyncio.run(
explore(
args.explore,
contract=exploration_contract,
bundle_dirs=(args.bundle_dir,),
profile=args.profile,
client_factory=scripted_client_factory,
trace=exploration_trace,
)
)
finally:
# From a ``finally``, exactly as ``write_parse_failures`` is (Fase 1b, funn 1): the run
# that most needs this evidence is the one a cap cut short, and that run returns
# nothing. ``completed`` says which of the two happened, so a reader never has to infer
# it from an absent ``stop``.
if args.outbox_dir and args.run_id:
outbox.write_exploration(
args.outbox_dir,
args.run_id,
payload=trace_payload(
exploration_trace,
stop=exploration.stop if exploration is not None else None,
completed=exploration is not None,
),
)
print(exploration_notice(exploration))
mandate = exploration.mandate
if mandate is not None: if mandate is not None:
# The scope line reads the dimension config only to NAME it. A config that fails to load is # The scope line reads the dimension config only to NAME it. A config that fails to load is
# left unnamed here and refused a moment later by the dispatch below, which stays the single # left unnamed here and refused a moment later by the dispatch below, which stays the single

View file

@ -0,0 +1,538 @@
"""U4 + U13, part 2 — the CALL SITES. Load-bearing proofs for the seams econ 56 left open.
Three things are proved here, and each of them is a seam a mutation can detach.
**1. The trace is a CALLER-OWNED accumulator, for the reason the parse-failure sink is one
(Fase 1b, funn 1).** ``explore()`` raises ``BudgetExceeded`` on its round cap, and a token cap
fires from inside the middleware mid-run on both paths ``ExplorationResult`` never returns, so a
``ledger_log`` that existed only as a return value would be destroyed by exactly the endings § C.2
requires the artefact to be readable after ("så en stoppet utforskning er lesbar uansett hvilken
vakt som fyrte"). The accumulator the caller holds survives however the loop ended.
``ExplorationResult.ledger_log`` is BUILT FROM that accumulator rather than alongside it: two lists
holding one fact is the -(p) drift class, one layer up.
**2. The CLI door refuses everything it cannot honour, by name.** ``--explore`` and ``--mandate``
are two sources of ONE mandate and are REFUSED together rather than merged: ``explore()`` sets the
objective from the prompt and hardcodes ``allow_own_proposals=True``, so composing them would
silently overwrite three fields an operator wrote by hand. The refusal names the library API
(``explore(seed_approaches=)``) because § C.6 door 1 is a real need this surface does not serve.
**3. ``{run_id}-exploration.json`` is written from a ``finally``**, so the run that most needs the
evidence the one a cap cut short is the one that has it.
The client is the repo's own ``ScriptedChatClient`` throughout (a bare ``BaseChatClient`` no-ops
``BudgetMiddleware``), and every tool assertion calls the tool's ``func`` DIRECTLY: measured in
econ 56, a scripted run returns TEXT and never emits a tool call, so no scripted exploration
reaches a tool body and a gate that only drove ``explore()`` would be vacuous.
"""
from __future__ import annotations
import json
from collections.abc import Callable
from pathlib import Path
from typing import Any
import pytest
from agent_framework import BaseChatClient
import portfolio_optimiser
from portfolio_optimiser import explore, okf, run
from portfolio_optimiser.budget import BudgetExceeded
from portfolio_optimiser.explore import ExplorationContract, ExplorationTrace
from portfolio_optimiser.mandate import Approach, Mandate
from portfolio_optimiser.simulation import ScriptedChatClient
_BUNDLE_DIR = Path(__file__).resolve().parents[1] / "shared" / "examples" / "bygg-energi-mikro"
_PID = "BYGG-KONTOR-NORD"
#: The hypothesiser's marked line. The label is the marker the end-to-end arm looks for on stdout:
#: it appears nowhere in the bundle, in the reference projects or in any other test, so its presence
#: in the settlement can only have come through the mandate the exploration shaped.
_LABEL = "SENTINEL-EXPLORE-7c1d33"
_ENERGY_REPLY = (
'{"measure":"LED-retrofit av kontorbelysning","affected_items":'
'[{"code":"ENERGI-TOTAL-EL","quantity":300000,"unit_cost":1.0}],"claimed_saving_nok":30000}'
)
_CONTRACT_JSON: dict[str, Any] = {
"max_rounds": 4,
"max_tokens": 100_000,
"max_stall_count": 2,
"max_reset_count": 1,
"max_plan_revisions": 0,
"enable_plan_review": False,
}
@pytest.fixture(autouse=True)
def _isolate_model_env(monkeypatch: pytest.MonkeyPatch) -> None:
"""Hermetic env: no arm here may read the operator's Foundry configuration."""
monkeypatch.delenv("PORTFOLIO_MODEL_MAP", raising=False)
monkeypatch.delenv("PORTFOLIO_FOUNDRY_PROJECT_ENDPOINT", raising=False)
def _ledger_json(*, satisfied: bool, speaker: str = "hypothesiser") -> str:
return json.dumps(
{
"is_request_satisfied": {"reason": "r", "answer": satisfied},
"is_in_loop": {"reason": "r", "answer": False},
"is_progress_being_made": {"reason": "r", "answer": True},
"next_speaker": {"reason": "r", "answer": speaker},
"instruction_or_question": {"reason": "r", "answer": "Shape one hypothesis."},
}
)
def _manager_script(ledgers: list[str]) -> Callable[[str, str], str]:
"""Route a manager prompt blob to its scripted reply (the econ-56 helper, verbatim in shape).
The stage ORDER is load-bearing (§ F, A6): the selector sees the CONCATENATION of the call's
messages, so a later-stage prompt still carries the earlier stage's text.
"""
def _select(blob: str, _role: str) -> str:
if "provide the final answer" in blob:
return "FINAL: exploration done."
if "pure JSON format" in blob:
return ledgers.pop(0) if ledgers else _ledger_json(satisfied=True)
if "went wrong on this last run" in blob:
return "PLAN-UPDATE: revised plan."
if "rewrite the following fact sheet" in blob:
return "FACTS-UPDATE: revised facts."
if "bullet-point plan" in blob:
return "PLAN: - ask the hypothesiser"
if "pre-survey" in blob:
return "FACTS: the bundle is anchored."
return "{}"
return _select
def _hypothesis_line(label: str, rationale: str) -> str:
return f"{explore.HYPOTHESIS_MARKER} " + json.dumps({"label": label, "rationale": rationale})
def _factory(
*, ledgers: list[str], hypothesiser: list[str], fallback: str = "ok"
) -> Callable[[str], BaseChatClient]:
"""One fresh ``ScriptedChatClient`` per role — the exploration's three plus everyone else.
``fallback`` serves the roles the PIPELINE builds (proposer/checker), so one factory can drive
an exploration and the run it hands its mandate to. That is what makes the end-to-end arm an
end-to-end arm rather than two half-proofs.
"""
def factory(role: str) -> BaseChatClient:
if role == explore.MANAGER_ROLE:
return ScriptedChatClient(reply_selector=_manager_script(ledgers), role=role)
if role == explore.HYPOTHESISER_ROLE:
replies = list(hypothesiser)
def _hyp(_blob: str, _role: str) -> str:
return replies.pop(0) if replies else "nothing further."
return ScriptedChatClient(reply_selector=_hyp, role=role)
if role == explore.NAVIGATOR_ROLE:
return ScriptedChatClient("NAVIGATOR: index read.", role=role)
return ScriptedChatClient(fallback, role=role)
return factory
def _contract(**overrides: Any) -> ExplorationContract:
return ExplorationContract(**{**_CONTRACT_JSON, **overrides})
# ---------------------------------------------------------------------------------------------
# 1. The caller-owned accumulator (the funn-1 sink shape, applied to the exploration)
# ---------------------------------------------------------------------------------------------
def _micro_bundle_dir() -> str:
return str(
Path(portfolio_optimiser.__file__).parent
/ "data"
/ "bundles"
/ "bygg-energi-baseline-mikro"
)
def test_quick_validate_verdicts_reach_the_callers_trace() -> None:
"""T1: every advisory verdict the hypothesiser asked for is recorded where the caller can read
it the ONE thing ``ExplorationResult`` deliberately does not carry.
Called DIRECTLY, because a scripted run never reaches a tool body (measured, econ 56): an arm
that drove ``explore()`` and then asserted on an empty list would be green against every
implementation, including one with no sink at all.
Detach point: drop the ``sink`` append in ``quick_validate_tool`` RED.
"""
base = _micro_bundle_dir()
projection = dict(okf.load_ir_projection(base))
projection.pop("_note", None)
trace = ExplorationTrace()
validate = explore.quick_validate_tool((base,), sink=trace.quick_validations)
honest = validate.func(
bundle_id="bygg-energi-baseline-mikro", proposal_json=json.dumps(projection)
)
invented = dict(projection)
invented["affected_items"] = [
{**dict(projection["affected_items"][0]), "code": "CODE-THAT-DOES-NOT-EXIST"}
]
validate.func(bundle_id="bygg-energi-baseline-mikro", proposal_json=json.dumps(invented))
assert len(trace.quick_validations) == 2, "both calls must be recorded, in call order"
first, second = trace.quick_validations
assert first.bundle_id == "bygg-energi-baseline-mikro"
assert first.verdict == honest, "the recorded verdict must be the one the tool ANSWERED"
assert second.verdict["decision"] == "rejected"
assert "CODE-THAT-DOES-NOT-EXIST" in second.proposal_json
@pytest.mark.asyncio
async def test_the_returned_ledger_log_is_the_traces_own_entries() -> None:
"""T2: ``ExplorationResult.ledger_log`` is BUILT FROM the accumulator, never alongside it.
Two lists holding one fact drift (-(p)), and a drifted pair would let the returned result and
the written artefact describe different runs.
Detach point: accumulate rounds in a second local list RED.
"""
trace = ExplorationTrace()
result = await explore.explore(
"Find a saving.",
contract=_contract(),
bundle_dirs=(str(_BUNDLE_DIR),),
client_factory=_factory(
ledgers=[_ledger_json(satisfied=False), _ledger_json(satisfied=True)],
hypothesiser=[_hypothesis_line(_LABEL, "because the bundle says so")],
),
trace=trace,
)
assert result.stop is None
assert len(trace.ledger) == 2
assert tuple(trace.ledger) == result.ledger_log
@pytest.mark.asyncio
async def test_the_trace_survives_the_budget_exception_that_destroys_the_result() -> None:
"""T3: the round cap raises, and the caller STILL holds every round the loop recorded.
This is the whole reason the accumulator is caller-owned. The round cap leaves as a typed
``BudgetExceeded`` (econ 56), so nothing is returned and § C.2 requires the artefact to be
readable no matter which guard fired.
Detach point: return the log only, keeping no caller-visible accumulator RED.
"""
trace = ExplorationTrace()
with pytest.raises(BudgetExceeded) as excinfo:
await explore.explore(
"Find a saving.",
contract=_contract(max_rounds=2),
bundle_dirs=(str(_BUNDLE_DIR),),
client_factory=_factory(
ledgers=[_ledger_json(satisfied=False), _ledger_json(satisfied=False)],
hypothesiser=["still thinking."],
),
trace=trace,
)
assert excinfo.value.kind == "exploration_rounds"
assert len(trace.ledger) == 2, (
"the rounds the exploration DID record were destroyed with the result — the artefact a "
"capped run needs most would be empty"
)
# ---------------------------------------------------------------------------------------------
# 2. The CLI door — every refusal by name, never a silent merge or a silent drop
# ---------------------------------------------------------------------------------------------
def _config_file(tmp_path: Path, **overrides: Any) -> str:
path = tmp_path / "exploration.json"
path.write_text(json.dumps({**_CONTRACT_JSON, **overrides}), encoding="utf-8")
return str(path)
def _mandate_file(tmp_path: Path) -> str:
path = tmp_path / "mandate.json"
path.write_text(
json.dumps(
{
"objective": "cut energy cost",
"approaches": [{"id": "a1", "label": "LED", "description": "swap the fittings"}],
"allow_own_proposals": False,
}
),
encoding="utf-8",
)
return str(path)
def _base_argv(tmp_path: Path) -> list[str]:
return [
_PID,
"--docs-dir",
str(_BUNDLE_DIR),
"--bundle-dir",
str(_BUNDLE_DIR),
"--explore",
"Find the cheapest saving.",
"--explore-config",
_config_file(tmp_path),
]
def test_an_exploration_config_without_an_exploration_is_refused_by_name(tmp_path, capsys) -> None:
"""T4: ``--explore-config`` alone would be loaded and then dropped on the floor — the exact
silent-ignore ``--embedder-config requires --semantic-retrieval`` exists to prevent.
Detach point: drop the refusal RED.
"""
rc = run.main(
[_PID, "--docs-dir", str(_BUNDLE_DIR), "--explore-config", _config_file(tmp_path)]
)
assert rc == 1
assert "--explore-config" in capsys.readouterr().err
def test_an_exploration_without_its_bounds_is_refused_rather_than_defaulted(
tmp_path, capsys
) -> None:
"""T5: ``--explore`` alone is refused — the CLI may not invent bounds.
Every ``ExplorationContract`` field is required WITHOUT a default precisely because
``MagenticBuilder`` falls back to unbounded, and a CLI that supplied its own numbers would undo
that decision one layer up.
"""
rc = run.main(
[_PID, "--docs-dir", str(_BUNDLE_DIR), "--bundle-dir", str(_BUNDLE_DIR), "--explore", "go"]
)
assert rc == 1
assert "--explore-config" in capsys.readouterr().err
def test_explore_and_mandate_are_two_sources_of_one_mandate_and_are_refused_together(
tmp_path, capsys
) -> None:
"""T6: the decision, made deliberately and stated: REFUSE, never merge.
``explore()`` takes the objective from the prompt and hardcodes ``allow_own_proposals=True``, so
composing the two would silently overwrite fields the operator wrote by hand. The message names
the library door (``seed_approaches``) so the refusal teaches instead of only forbidding.
Detach point: let one source silently win RED.
"""
rc = run.main(_base_argv(tmp_path) + ["--mandate", _mandate_file(tmp_path)])
assert rc == 1
err = capsys.readouterr().err
assert "--explore" in err and "--mandate" in err
assert "seed_approaches" in err, "the refusal must name the door that DOES serve door 1"
def test_explore_and_live_dry_run_contradict_and_are_refused(tmp_path, capsys) -> None:
"""T7: ``--live-dry-run`` stops before the first model call; an exploration IS model calls."""
rc = run.main(_base_argv(tmp_path) + ["--live-dry-run"])
assert rc == 1
assert "--live-dry-run" in capsys.readouterr().err
def test_an_exploration_with_no_knowledge_base_is_refused(tmp_path, capsys) -> None:
"""T8: without ``--bundle-dir`` the navigator has nothing to open — the loop would run, cost
tokens and read nothing. Refused rather than run empty (the ``--semantic-retrieval`` shape).
Detach point: drop the requirement RED.
"""
rc = run.main(
[
_PID,
"--docs-dir",
str(_BUNDLE_DIR),
"--explore",
"go",
"--explore-config",
_config_file(tmp_path),
]
)
assert rc == 1
assert "--bundle-dir" in capsys.readouterr().err
def test_a_plan_review_nobody_can_answer_is_refused_at_the_cli(tmp_path, capsys) -> None:
"""T9: ``enable_plan_review`` is the U13 SYNCHRONOUS door and this surface has no reviewer.
Refused HERE rather than left to ``explore()``: ``ExplorationError`` is a ``RuntimeError``, so
it is outside ``main()``'s ``(ValueError, FileNotFoundError, ValidationError)`` refusal tuple
and would leave as a traceback instead of the rc-1 line every other misconfiguration produces.
Detach point: let the flag through to ``explore()`` RED (traceback, not rc 1).
"""
rc = run.main(
_base_argv(tmp_path)[:-1] + [_config_file(tmp_path, enable_plan_review=True)],
)
assert rc == 1
assert "enable_plan_review" in capsys.readouterr().err
def test_explore_belongs_to_single_project_mode(tmp_path, capsys) -> None:
"""T10: portfolio mode is a documented partition, and ``--explore`` is on the single-project
side of it one exploration shapes ONE mandate against ONE knowledge base.
The assertion names ``--portfolio``, and that was MEASURED rather than chosen: asserting only
that the message mentions ``--explore`` passed against an implementation with no partition
entry at all, because the run then fell through to ``--explore requires --bundle-dir``, which
names ``--explore`` too. Two refusals sharing a substring is this repo's "assert never on
wording two branches share" rule, caught by its own mutation.
Detach point: drop ``--explore`` from the portfolio ``single_only`` partition RED.
"""
rc = run.main(["--portfolio", "--explore", "go", "--explore-config", _config_file(tmp_path)])
assert rc == 1
err = capsys.readouterr().err
assert "--explore" in err and "--portfolio" in err
@pytest.fixture()
def _explored_main(monkeypatch: pytest.MonkeyPatch) -> None:
"""Inject the role-dispatching scripted factory into the seam ``main()`` resolves through.
``main()`` passes no ``client_factory``, and ``explore()`` imports ``run._default_factory``
lazily at call time, so this ONE patch covers both the exploration and the pipeline it feeds
which is what makes the arm below end-to-end rather than a wiring spy.
"""
factory = _factory(
ledgers=[_ledger_json(satisfied=False), _ledger_json(satisfied=True)],
hypothesiser=[_hypothesis_line(_LABEL, "the index says the fittings are old")],
fallback=_ENERGY_REPLY,
)
monkeypatch.setattr("portfolio_optimiser.run._default_factory", lambda profile: factory)
def test_the_shaped_mandate_reaches_the_pipeline(tmp_path, capsys, _explored_main) -> None:
"""T11: the approach the hypothesiser shaped is SETTLED by the run — the whole point of (1).
The settlement is printed only for a run that HAS a mandate, and the label appears nowhere in
the bundle or the reference projects, so it can have reached stdout only by travelling
prompt ``explore()`` ``Mandate`` ``run_project(mandate=)`` ``settle``.
Detach point: drop ``mandate=`` from the exploring branch's ``run_project`` call → RED.
"""
rc = run.main(_base_argv(tmp_path))
assert rc == 0
out = capsys.readouterr().out
assert _LABEL in out, "the exploration's mandate never reached the pipeline's settlement"
# ---------------------------------------------------------------------------------------------
# 3. The artefact — written from a ``finally``, because a capped run is what it exists for
# ---------------------------------------------------------------------------------------------
def test_the_exploration_artefact_carries_the_rounds_and_the_advisory_verdicts(
tmp_path, _explored_main
) -> None:
"""T12: ``{run_id}-exploration.json`` holds the per-round ledger AND the ``quick_validate``
verdicts the level-1 evidence ``ExplorationResult`` deliberately does not carry (§ C.2).
Detach point: drop the artefact write RED.
"""
outbox = tmp_path / "outbox"
rc = run.main(_base_argv(tmp_path) + ["--outbox-dir", str(outbox), "--run-id", "r1"])
assert rc == 0
payload = json.loads((outbox / "r1-exploration.json").read_text(encoding="utf-8"))
assert payload["run_id"] == "r1"
assert payload["completed"] is True
assert payload["stop"] is None
assert [row["round_index"] for row in payload["rounds"]] == [1, 2]
assert payload["rounds"][-1]["is_request_satisfied"] is True
assert payload["rounds"][0]["next_speaker"] == "hypothesiser"
assert "quick_validations" in payload
def test_the_artefact_is_written_even_when_the_exploration_was_cut_short(
tmp_path, monkeypatch
) -> None:
"""T13: a capped exploration is the run whose evidence matters MOST, and it is the one that
returns nothing so the write lives in a ``finally`` (the ``write_parse_failures`` precedent).
``completed`` is a field rather than an inference: with no result there is no ``stop``, and a
``stop: null`` that meant BOTH "concluded normally" and "we never found out" would be the kind
of silence this repo writes required fields to close.
Detach point: move the write out of the ``finally`` RED.
"""
factory = _factory(
ledgers=[_ledger_json(satisfied=False), _ledger_json(satisfied=False)],
hypothesiser=["still thinking."],
fallback=_ENERGY_REPLY,
)
monkeypatch.setattr("portfolio_optimiser.run._default_factory", lambda profile: factory)
outbox = tmp_path / "outbox"
with pytest.raises(BudgetExceeded):
run.main(
[
_PID,
"--docs-dir",
str(_BUNDLE_DIR),
"--bundle-dir",
str(_BUNDLE_DIR),
"--explore",
"go",
"--explore-config",
_config_file(tmp_path, max_rounds=2),
"--outbox-dir",
str(outbox),
"--run-id",
"r2",
]
)
payload = json.loads((outbox / "r2-exploration.json").read_text(encoding="utf-8"))
assert payload["completed"] is False
assert payload["stop"] is None
assert len(payload["rounds"]) == 2
def test_the_artefact_payload_is_byte_deterministic() -> None:
"""T14 (control): the same trace renders the same bytes, so the artefact is diff-stable like
every other outbox file. Drives the renderer directly the CLI arms above prove it is CALLED,
this proves what it produces."""
trace = ExplorationTrace()
trace.ledger.append(
explore.LedgerEntry(
round_index=1,
is_request_satisfied=True,
is_in_loop=False,
is_progress_being_made=True,
next_speaker="hypothesiser",
instruction_or_question="Shape one hypothesis.",
speaker_known=True,
)
)
trace.quick_validations.append(
explore.QuickValidation(
bundle_id="b", proposal_json="{}", verdict={"decision": "unparseable"}
)
)
first = explore.trace_payload(trace, stop=None, completed=True)
second = explore.trace_payload(trace, stop=None, completed=True)
assert json.dumps(first, sort_keys=True) == json.dumps(second, sort_keys=True)
def test_a_seeded_mandate_still_leads_the_shaped_one() -> None:
"""T15 (control for T6's refusal): the library door the refusal names actually works.
A refusal that pointed at a door which did not open would be worse than no message at all.
"""
seed = Approach(id="expert-1", label="expert's own", description="the domain expert asked")
minted = explore._mint_approaches((seed,), [(_LABEL, "shaped in the loop")])
assert [a.id for a in minted] == ["expert-1", "hypothesis-1"]
assert isinstance(Mandate(objective="o", approaches=minted, allow_own_proposals=True), Mandate)