feat(toolbox): the seven outbox writers get their own door -- row 1 moves 8 -> 15 of 17
Every one of the run path's seven `outbox.write_*` steps was reachable only through `run.main`, and every path through that builds a chat client. The steps need no model: they take already-rendered data and put it on disk in a byte-deterministic form. Seven subcommands, seven thin adapters. The outbox directory is always the caller's to name -- never a default, never the repository's own, because a step that wrote into a folder the framework also reads as an inbox would bypass the Step-8 promotion gate. `write-outbox` is the one that is not purely mechanical: `outbox.write_outbox` branches on the outcome TYPE, so a door that took the outcome as an argument would let anyone author an outbox of claims and hand it to Step 8 as results. The door DERIVES it through `validate_proposal` -- the run path's own composition -- and a blocked proposal exits 3 with the artefacts still written, since that is where the rejection is recorded. `verdict_id` stays an argument: `verdict-key` already owns that minting. `--stop-reason` is required rather than defaulted to the empty string, inheriting the core writer's measured reason: "the run finished" and "we never found out" must not be the same value. Eight probes, each a subprocess with the subcommand in argv, each asserting on the FILE the command wrote. The ground truth is composed in the test -- the payload it wrote and counted itself, and the byte form the contract requires -- never `outbox._dump`, which would have measured the module against itself. The refusal arm carries its rc-0 control. Measured, own run of the B gate: row 1 8 of 17 -> 15 of 17, exit 1 unchanged, no other row moved. 0 chat-client names reachable from the toolbox (known-positive control: 24 in run.py). Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
parent
1111924000
commit
6375ce5af3
5 changed files with 691 additions and 30 deletions
|
|
@ -263,11 +263,14 @@
|
|||
"scope": "run_project"
|
||||
},
|
||||
"entry": {
|
||||
"kind": "console-script",
|
||||
"module": "run.py",
|
||||
"scope": "main"
|
||||
"kind": "subcommand",
|
||||
"module": "toolbox.py",
|
||||
"scope": "main",
|
||||
"command": "write-run-config"
|
||||
},
|
||||
"probe": []
|
||||
"probe": [
|
||||
"tests/test_toolbox_outbox_doors.py::test_write_run_config_from_outside_is_the_byte_deterministic_artefact"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "coverage",
|
||||
|
|
@ -279,11 +282,14 @@
|
|||
"scope": "run_project"
|
||||
},
|
||||
"entry": {
|
||||
"kind": "console-script",
|
||||
"module": "run.py",
|
||||
"scope": "main"
|
||||
"kind": "subcommand",
|
||||
"module": "toolbox.py",
|
||||
"scope": "main",
|
||||
"command": "write-coverage"
|
||||
},
|
||||
"probe": []
|
||||
"probe": [
|
||||
"tests/test_toolbox_outbox_doors.py::test_write_coverage_from_outside_keeps_every_row_and_the_stop_reason"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "utboks",
|
||||
|
|
@ -295,11 +301,14 @@
|
|||
"scope": "run_project"
|
||||
},
|
||||
"entry": {
|
||||
"kind": "console-script",
|
||||
"module": "run.py",
|
||||
"scope": "main"
|
||||
"kind": "subcommand",
|
||||
"module": "toolbox.py",
|
||||
"scope": "main",
|
||||
"command": "write-outbox"
|
||||
},
|
||||
"probe": []
|
||||
"probe": [
|
||||
"tests/test_toolbox_outbox_doors.py::test_write_outbox_from_outside_writes_the_pair_the_run_path_writes"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "rundebinding",
|
||||
|
|
@ -365,11 +374,14 @@
|
|||
"scope": "run_project"
|
||||
},
|
||||
"entry": {
|
||||
"kind": "console-script",
|
||||
"module": "run.py",
|
||||
"scope": "main"
|
||||
"kind": "subcommand",
|
||||
"module": "toolbox.py",
|
||||
"scope": "main",
|
||||
"command": "write-prepass"
|
||||
},
|
||||
"probe": []
|
||||
"probe": [
|
||||
"tests/test_toolbox_outbox_doors.py::test_write_prepass_from_outside_records_the_cut_the_run_was_given"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "parse-feil",
|
||||
|
|
@ -381,11 +393,14 @@
|
|||
"scope": "run_project"
|
||||
},
|
||||
"entry": {
|
||||
"kind": "console-script",
|
||||
"module": "run.py",
|
||||
"scope": "main"
|
||||
"kind": "subcommand",
|
||||
"module": "toolbox.py",
|
||||
"scope": "main",
|
||||
"command": "write-parse-failures"
|
||||
},
|
||||
"probe": []
|
||||
"probe": [
|
||||
"tests/test_toolbox_outbox_doors.py::test_write_parse_failures_from_outside_keeps_every_reply_that_did_not_parse"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "forslagsvurderinger",
|
||||
|
|
@ -397,11 +412,14 @@
|
|||
"scope": "run_project"
|
||||
},
|
||||
"entry": {
|
||||
"kind": "console-script",
|
||||
"module": "run.py",
|
||||
"scope": "main"
|
||||
"kind": "subcommand",
|
||||
"module": "toolbox.py",
|
||||
"scope": "main",
|
||||
"command": "write-proposal-reviews"
|
||||
},
|
||||
"probe": []
|
||||
"probe": [
|
||||
"tests/test_toolbox_outbox_doors.py::test_write_proposal_reviews_from_outside_states_an_empty_review_list"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "debatt-verktøy",
|
||||
|
|
@ -413,11 +431,14 @@
|
|||
"scope": "run_project"
|
||||
},
|
||||
"entry": {
|
||||
"kind": "console-script",
|
||||
"module": "run.py",
|
||||
"scope": "main"
|
||||
"kind": "subcommand",
|
||||
"module": "toolbox.py",
|
||||
"scope": "main",
|
||||
"command": "write-debate-tools"
|
||||
},
|
||||
"probe": []
|
||||
"probe": [
|
||||
"tests/test_toolbox_outbox_doors.py::test_write_debate_tools_from_outside_writes_an_empty_trace_as_a_statement"
|
||||
]
|
||||
}
|
||||
],
|
||||
"roles": {
|
||||
|
|
|
|||
|
|
@ -37,10 +37,11 @@ from dataclasses import dataclass
|
|||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from portfolio_optimiser import okf, prepass
|
||||
from portfolio_optimiser import okf, outbox, prepass
|
||||
from portfolio_optimiser.contracts import FeedbackContract
|
||||
from portfolio_optimiser.datasource import retrieve_chunks
|
||||
from portfolio_optimiser.ir import SavingsProposal
|
||||
from portfolio_optimiser.provenance import ProvenanceStamp
|
||||
from portfolio_optimiser.validator import (
|
||||
ValidatedProposal,
|
||||
rejection_stage,
|
||||
|
|
@ -255,6 +256,151 @@ def capture_verdict_command(args: argparse.Namespace) -> Mapping[str, Any]:
|
|||
return verdict_to_dict(verdict)
|
||||
|
||||
|
||||
def _payload_file(path: str, expected: type) -> Any:
|
||||
"""A JSON argument, read as the shape the core writer already takes.
|
||||
|
||||
The type is checked HERE rather than left to the writer, because the writers take
|
||||
``Mapping``/``Sequence`` and would serialize whatever they were handed: a list where an
|
||||
object belongs becomes a valid file with the wrong shape, and the run that reads it later is
|
||||
the one that fails. Exit 2 is the wrong code for it (the call parsed), so it is a refusal."""
|
||||
value = json.loads(Path(path).read_text(encoding="utf-8"))
|
||||
if not isinstance(value, expected):
|
||||
raise ValueError(f"{path}: expected a JSON {expected.__name__}, got {type(value).__name__}")
|
||||
return value
|
||||
|
||||
|
||||
def write_run_config_command(args: argparse.Namespace) -> Mapping[str, Any]:
|
||||
"""``write-run-config`` — everything that describes a run WITHOUT a model call.
|
||||
|
||||
The first artefact a caller weighing whether to pay for a debate has use for: the resolved
|
||||
deployment per role, the profile and the caps, in the byte-deterministic form the comparison
|
||||
protocol reads. ``resolved_models`` comes in as data because the run path resolves it before
|
||||
this step too (``resolve_model`` is held out of the toolbox for exactly that reason — it
|
||||
looks up the chat client's deployment)."""
|
||||
models = _payload_file(args.resolved_models, dict)
|
||||
path = outbox.write_run_config(
|
||||
args.out_dir,
|
||||
args.run_id,
|
||||
profile=args.profile,
|
||||
resolved_models={str(role): str(model) for role, model in models.items()},
|
||||
max_rounds=args.max_rounds,
|
||||
max_tokens=args.max_tokens,
|
||||
top_k=args.top_k,
|
||||
)
|
||||
return {"path": str(path), "run_id": args.run_id}
|
||||
|
||||
|
||||
def write_coverage_command(args: argparse.Namespace) -> Mapping[str, Any]:
|
||||
"""``write-coverage`` — WHY each commissioned approach ended as it did.
|
||||
|
||||
``--stop-reason`` is required, never defaulted to the empty string, because the core writer
|
||||
requires it for a measured reason: "the run finished" and "we never found out" must not be
|
||||
the same value. A door that supplied the empty string on the caller's behalf would turn
|
||||
every unfinished run into a finished one."""
|
||||
rows = _payload_file(args.rows, list)
|
||||
path = outbox.write_coverage(
|
||||
args.outbox_dir, args.run_id, rows=rows, stop_reason=args.stop_reason
|
||||
)
|
||||
return {"path": str(path), "run_id": args.run_id, "rows": len(rows)}
|
||||
|
||||
|
||||
def write_outbox_command(args: argparse.Namespace) -> Mapping[str, Any] | Refused:
|
||||
"""``write-outbox`` — the proposal/outcome pair, with the outcome DERIVED, not declared.
|
||||
|
||||
The outcome is not an argument, and that is the load-bearing decision here. ``write_outbox``
|
||||
branches on the outcome TYPE — a ``ValidatedProposal`` writes percentiles, a ``Rejection``
|
||||
writes its reason — so a door that let the caller hand in either would let anyone author an
|
||||
outbox of claims and hand it to Step 8 as results. The door runs the proposal through the
|
||||
same blocking gate the run path runs it through, and writes what came back.
|
||||
|
||||
``verdict_id`` IS an argument: the run path mints it with ``verdict_key`` before this step,
|
||||
and that name has its own door (``verdict-key``). Minting it a second time here would give
|
||||
the learning key two producers.
|
||||
|
||||
A blocked proposal is ``REFUSED``, and the artefacts are still written — they are where the
|
||||
rejection is recorded. The exit code answers the caller's question, not the writer's: an
|
||||
agent that only reads rc must never take a blocked proposal for a cleared one."""
|
||||
proposal = _proposal_from(args.proposal)
|
||||
stamp = ProvenanceStamp.model_validate_json(Path(args.provenance).read_text(encoding="utf-8"))
|
||||
baseline = (
|
||||
None if args.cost_baseline is None else okf.load_cost_baseline_file(args.cost_baseline)
|
||||
)
|
||||
outcome = validate_proposal(proposal, baseline=baseline)
|
||||
proposal_path, outcome_path = outbox.write_outbox(
|
||||
args.outbox_dir,
|
||||
args.run_id,
|
||||
outcome=outcome,
|
||||
provenance=stamp,
|
||||
checker_verdict=args.checker_verdict,
|
||||
verdict_id=args.verdict_id,
|
||||
approach_id=args.approach_id,
|
||||
)
|
||||
validated = isinstance(outcome, ValidatedProposal)
|
||||
payload = {
|
||||
"decision": "validated" if validated else "rejected",
|
||||
"proposal_path": str(proposal_path),
|
||||
"outcome_path": str(outcome_path),
|
||||
"run_id": args.run_id,
|
||||
"verdict_id": args.verdict_id,
|
||||
}
|
||||
return payload if validated else Refused(payload)
|
||||
|
||||
|
||||
def write_prepass_command(args: argparse.Namespace) -> Mapping[str, Any]:
|
||||
"""``write-prepass`` — the CUT this run was given, as a file.
|
||||
|
||||
Its PRESENCE is what tells "withdrawn by design" apart from the S2c regression, so the door
|
||||
writes whenever it is called, exactly as the run path writes whenever a payload was given."""
|
||||
declaration = _payload_file(args.declaration, dict)
|
||||
path = outbox.write_prepass(args.outbox_dir, args.run_id, declaration=declaration)
|
||||
return {"path": str(path), "run_id": args.run_id}
|
||||
|
||||
|
||||
def write_parse_failures_command(args: argparse.Namespace) -> Mapping[str, Any]:
|
||||
"""``write-parse-failures`` — the replies that did NOT become typed IR.
|
||||
|
||||
Plain mappings in, exactly as the run path flattens ``generate.ParseFailure`` before this
|
||||
step: the RAW output layer stays free of the module that imports MAF, and so does this door."""
|
||||
failures = _payload_file(args.failures, list)
|
||||
path = outbox.write_parse_failures(
|
||||
args.outbox_dir,
|
||||
args.run_id,
|
||||
failures=[{str(k): str(v) for k, v in f.items()} for f in failures],
|
||||
)
|
||||
return {"path": str(path), "run_id": args.run_id, "failures": len(failures)}
|
||||
|
||||
|
||||
def write_proposal_reviews_command(args: argparse.Namespace) -> Mapping[str, Any]:
|
||||
"""``write-proposal-reviews`` — what a human answered about the proposals on the table.
|
||||
|
||||
Already-rendered payload in, for the run path's reason: the renderer lives in the module that
|
||||
owns the type. The write rule — iff a reviewer was given, INCLUDING an empty list — is the
|
||||
caller's here, because calling this door IS giving one."""
|
||||
payload = _payload_file(args.payload, dict)
|
||||
path = outbox.write_proposal_reviews(args.outbox_dir, args.run_id, payload=payload)
|
||||
return {"path": str(path), "run_id": args.run_id}
|
||||
|
||||
|
||||
def write_debate_tools_command(args: argparse.Namespace) -> Mapping[str, Any]:
|
||||
"""``write-debate-tools`` — WHICH documents were opened, and WHICH requirement was binding.
|
||||
|
||||
``--requirements`` defaults to empty for the reason the core writer's parameter does: "this
|
||||
debate declared nothing" is an honest positive statement, and an empty trace is the S2c
|
||||
regression itself — it has to be readable off the artefact, never inferred from a file that
|
||||
is not there."""
|
||||
tool_calls = _payload_file(args.tool_calls, list)
|
||||
requirements = [] if args.requirements is None else _payload_file(args.requirements, list)
|
||||
path = outbox.write_debate_tools(
|
||||
args.outbox_dir, args.run_id, tool_calls=tool_calls, requirements=requirements
|
||||
)
|
||||
return {
|
||||
"path": str(path),
|
||||
"run_id": args.run_id,
|
||||
"tool_calls": len(tool_calls),
|
||||
"requirements": len(requirements),
|
||||
}
|
||||
|
||||
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
"""The doors, each registered by name.
|
||||
|
||||
|
|
@ -303,6 +449,69 @@ def build_parser() -> argparse.ArgumentParser:
|
|||
fang.add_argument("--proposal", required=True, help="the proposal IR as JSON")
|
||||
fang.add_argument("--decision", required=True, help="the expert's decision")
|
||||
fang.add_argument("--rationale", required=True, help="why — carried into the store verbatim")
|
||||
|
||||
#: The outbox doors. Each takes an outbox directory FROM THE CALLER — never a default, and
|
||||
#: never the repository's own: a step that wrote into a folder the framework also reads as an
|
||||
#: inbox would bypass the Step-8 promotion gate, and a probe that did it would leave files a
|
||||
#: later run counts as its own.
|
||||
konfig = sub.add_parser("write-run-config", help="write the run-config artefact")
|
||||
konfig.add_argument("--out-dir", required=True, help="where the artefact is written")
|
||||
konfig.add_argument("--run-id", required=True, help="the run the artefact belongs to")
|
||||
konfig.add_argument("--profile", required=True, help="the backend profile the run used")
|
||||
konfig.add_argument(
|
||||
"--resolved-models", required=True, help="JSON object: role -> resolved deployment"
|
||||
)
|
||||
konfig.add_argument("--max-rounds", type=int, required=True, help="the round cap")
|
||||
konfig.add_argument("--max-tokens", type=int, required=True, help="the token cap")
|
||||
konfig.add_argument("--top-k", type=int, required=True, help="retrieval depth")
|
||||
|
||||
dekning = sub.add_parser("write-coverage", help="write the approach-coverage artefact")
|
||||
dekning.add_argument("--outbox-dir", required=True, help="where the artefact is written")
|
||||
dekning.add_argument("--run-id", required=True, help="the run the artefact belongs to")
|
||||
dekning.add_argument("--rows", required=True, help="JSON array of coverage rows")
|
||||
dekning.add_argument(
|
||||
"--stop-reason",
|
||||
required=True,
|
||||
help="what cut the run short, or the empty string when nothing did — required, because "
|
||||
"'the run finished' and 'we never found out' must not be the same value",
|
||||
)
|
||||
|
||||
utboks = sub.add_parser("write-outbox", help="write the proposal/outcome artefact pair")
|
||||
utboks.add_argument("--outbox-dir", required=True, help="where the artefacts are written")
|
||||
utboks.add_argument("--run-id", required=True, help="the run the artefacts belong to")
|
||||
utboks.add_argument("--proposal", required=True, help="the proposal IR as JSON")
|
||||
utboks.add_argument("--provenance", required=True, help="the provenance stamp as JSON")
|
||||
utboks.add_argument("--verdict-id", required=True, help="the key from verdict-key")
|
||||
utboks.add_argument("--checker-verdict", default=None, help="the checker's decision, if any")
|
||||
utboks.add_argument("--approach-id", default=None, help="which commissioned approach this is")
|
||||
utboks.add_argument(
|
||||
"--cost-baseline",
|
||||
default=None,
|
||||
help="the project's own priced lines; without it stage 0 never runs",
|
||||
)
|
||||
|
||||
kutt = sub.add_parser("write-prepass", help="write the pre-pass declaration artefact")
|
||||
kutt.add_argument("--outbox-dir", required=True, help="where the artefact is written")
|
||||
kutt.add_argument("--run-id", required=True, help="the run the artefact belongs to")
|
||||
kutt.add_argument("--declaration", required=True, help="the producer's declaration JSON")
|
||||
|
||||
parse = sub.add_parser("write-parse-failures", help="write the unparsed-replies artefact")
|
||||
parse.add_argument("--outbox-dir", required=True, help="where the artefact is written")
|
||||
parse.add_argument("--run-id", required=True, help="the run the artefact belongs to")
|
||||
parse.add_argument("--failures", required=True, help="JSON array of {text, error} objects")
|
||||
|
||||
vurdering = sub.add_parser("write-proposal-reviews", help="write the expert-review artefact")
|
||||
vurdering.add_argument("--outbox-dir", required=True, help="where the artefact is written")
|
||||
vurdering.add_argument("--run-id", required=True, help="the run the artefact belongs to")
|
||||
vurdering.add_argument("--payload", required=True, help="the rendered review payload JSON")
|
||||
|
||||
debatt = sub.add_parser("write-debate-tools", help="write the debate navigation artefact")
|
||||
debatt.add_argument("--outbox-dir", required=True, help="where the artefact is written")
|
||||
debatt.add_argument("--run-id", required=True, help="the run the artefact belongs to")
|
||||
debatt.add_argument("--tool-calls", required=True, help="JSON array of tool calls, in order")
|
||||
debatt.add_argument(
|
||||
"--requirements", default=None, help="JSON array of declared binding requirements"
|
||||
)
|
||||
return parser
|
||||
|
||||
|
||||
|
|
@ -327,6 +536,20 @@ def dispatch(args: argparse.Namespace) -> Any:
|
|||
return verdict_key_command(args)
|
||||
if args.command == "capture-verdict":
|
||||
return capture_verdict_command(args)
|
||||
if args.command == "write-run-config":
|
||||
return write_run_config_command(args)
|
||||
if args.command == "write-coverage":
|
||||
return write_coverage_command(args)
|
||||
if args.command == "write-outbox":
|
||||
return write_outbox_command(args)
|
||||
if args.command == "write-prepass":
|
||||
return write_prepass_command(args)
|
||||
if args.command == "write-parse-failures":
|
||||
return write_parse_failures_command(args)
|
||||
if args.command == "write-proposal-reviews":
|
||||
return write_proposal_reviews_command(args)
|
||||
if args.command == "write-debate-tools":
|
||||
return write_debate_tools_command(args)
|
||||
raise RuntimeError(f"unregistered command {args.command!r}") # pragma: no cover - argparse
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue