[skip-docs] — the invariant row for this plan lands in Step 13, after the mutations. Amendment 2 in the same commit: the framework-neutrality sweep covered only expert-reviewer, so the skill that arrived by subtree pull had no framework guard anywhere. GUARDED_SKILL_DIRS names both, and a fail-closed coverage arm turns red when a future pull brings a third skill that is not listed. Co-Authored-By: Claude <claude-opus-5>
155 lines
8.2 KiB
Python
155 lines
8.2 KiB
Python
"""Step 12 (Session 5) - the falsification method as a SHARED Agent Skill, authored in commons.
|
|
|
|
**Operator decision D4, and why the skill is not written here.** ``shared/`` is a PULL-ONLY git
|
|
subtree of ``portfolio-optimiser-commons``. Authoring the skill locally would produce a wheel the
|
|
sibling never inherits, and **no test in this suite detects that divergence** --
|
|
``tests/test_shared_packaged_data_loadbearing.py`` compares the working tree against the wheel,
|
|
never against commons. The natural "fix" is ``git subtree push``, which leaked consumer history
|
|
once already, on 2026-07-03. So the two files under ``shared/skills/falsification-reviewer/``
|
|
arrive by ``git subtree pull``, and nothing in this file writes there.
|
|
|
|
**What is guarded HERE, and what deliberately is not.** The framework-neutrality rule has ONE copy
|
|
and it is NOT in this file: ``test_method_spec_loadbearing.test_method_spec_is_framework_neutral``
|
|
sweeps every shared spec and every skill tree, and Amendment 2 extended it to cover this skill.
|
|
Two copies of one rule is the ko-(p) drift class, and the plan concedes the framework guard is
|
|
close to trivially green on pure prose anyway. **The TERMINOLOGY guard is the one that bites**, and
|
|
it lives here: the customer-facing wording is "knowledge base" / "knowledge bundle", never the
|
|
internal format name, so the literal string ``OKF bundle`` appearing in the skill's prose turns
|
|
this file red.
|
|
|
|
**Amendment 3 decides the round trip's input, and it is not a preference.** ``frontmatter_verbatim``
|
|
is AUTHORITATIVE: it is the bytes a test materialises into a throwaway concept file before reading
|
|
them back. The sibling ``frontmatter`` object is a line-oriented PROJECTION and informative only --
|
|
by construction it cannot carry a block form, which is why concept 2 has no ``sources`` key there.
|
|
Materialising from ``frontmatter`` would derive ``state: absent`` and silently lose the
|
|
``unreadable`` case the example exists to demonstrate: the arm would be GREEN while proving the
|
|
opposite of its docstring. The example's own judgement is ``undecided`` -- not ``survived`` -- for
|
|
the same reason, because a claim whose one possible refuter was unreadable was never attacked.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from portfolio_optimiser import okf, persona
|
|
|
|
REPO_ROOT = Path(__file__).resolve().parents[1]
|
|
SKILL_DIR = REPO_ROOT / "shared" / "skills" / "falsification-reviewer"
|
|
EXAMPLE = SKILL_DIR / "references" / "example-evidence.json"
|
|
|
|
#: The internal format name. Customer-facing prose says "knowledge base" / "knowledge bundle".
|
|
_INTERNAL_NAME = re.compile(r"OKF bundle")
|
|
|
|
|
|
def test_the_skill_ships_its_two_files_with_usable_frontmatter() -> None:
|
|
"""(a) Structure. RED before the subtree pull has landed -- which is exactly Step 12's
|
|
On-failure clause: if the files are absent, STOP; do not author under ``shared/`` locally."""
|
|
skill_md = SKILL_DIR / "SKILL.md"
|
|
assert skill_md.is_file(), "shared/skills/falsification-reviewer/SKILL.md missing"
|
|
assert EXAMPLE.is_file(), "the skill's worked example is missing"
|
|
|
|
fm = okf.parse_frontmatter(skill_md)
|
|
assert fm.get("name") == "falsification-reviewer"
|
|
assert fm.get("description", "").strip(), "SKILL.md description must be non-empty"
|
|
json.loads(EXAMPLE.read_text(encoding="utf-8"))
|
|
|
|
|
|
def test_the_skill_never_uses_the_internal_format_name() -> None:
|
|
"""(b) The DISCRIMINATING guard. Paired with a KNOWN-POSITIVE control: a guard that searched
|
|
for something no file could ever contain would be green by construction, so the same regex is
|
|
first shown to fire on text that does carry the string."""
|
|
assert _INTERNAL_NAME.search("read the OKF bundle at that path") is not None, (
|
|
"the guard cannot find the string it is supposed to forbid -- it proves nothing"
|
|
)
|
|
for f in sorted(p for p in SKILL_DIR.rglob("*") if p.is_file()):
|
|
hit = _INTERNAL_NAME.search(f.read_text(encoding="utf-8"))
|
|
assert hit is None, f"internal format name {hit.group(0)!r} in customer-facing prose: {f}"
|
|
|
|
|
|
def test_the_example_resolves_at_call_time_not_at_import(
|
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
"""(c) The loader seam, mirroring ``persona.load_persona_example``: resolved INSIDE the call
|
|
(never frozen into a default argument), so both the test seam and ``PORTFOLIO_SHARED_ROOT``
|
|
stay live. A loader that bound its path at import would answer from the tree that existed when
|
|
the module was first imported."""
|
|
real = persona.load_falsification_example()
|
|
assert real.judgement == "undecided"
|
|
|
|
stand_in = tmp_path / "example-evidence.json"
|
|
stand_in.write_text(
|
|
json.dumps(
|
|
{
|
|
"claim": "c",
|
|
"judgement": "survived",
|
|
"refuter": None,
|
|
"concepts": [],
|
|
}
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
monkeypatch.setattr(persona, "_FALSIFICATION_EXAMPLE_PATH", stand_in)
|
|
assert persona.load_falsification_example().judgement == "survived"
|
|
|
|
|
|
def test_a_missing_example_fails_fast_rather_than_degrading(
|
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
"""(d) The example is REQUIRED input (contrast the tolerant Step-7 inbox). A broken shared
|
|
artefact must surface, not pass silently."""
|
|
monkeypatch.setattr(persona, "_FALSIFICATION_EXAMPLE_PATH", tmp_path / "nope.json")
|
|
with pytest.raises(FileNotFoundError):
|
|
persona.load_falsification_example()
|
|
|
|
|
|
def test_the_worked_example_round_trips_through_the_real_readers(tmp_path: Path) -> None:
|
|
"""(e) The example is not decoration: every concept's declared ``evidence`` and ``derived``
|
|
values are re-derived by the SHIPPED readers from the SHIPPED bytes.
|
|
|
|
``read_provenance`` takes a MARKDOWN path, not JSON, so the fields are materialised into a
|
|
throwaway concept file first (Amendment 3) -- ``frontmatter_verbatim`` VERBATIM, never
|
|
reassembled from the ``frontmatter`` projection. Reassembling would emit concept 2's
|
|
``sources`` as nothing at all, and the ``unreadable``/``block-sequence`` arm would pass as
|
|
``absent`` while claiming to prove the opposite.
|
|
|
|
Nothing git-tracked is touched: each concept is written under ``tmp_path``.
|
|
"""
|
|
example = persona.load_falsification_example()
|
|
assert len(example.concepts) == 2, "the example must keep both the readable and unreadable case"
|
|
|
|
seen_states = set()
|
|
for concept in example.concepts:
|
|
path = tmp_path / f"{concept.concept_id}.md"
|
|
path.write_text(concept.frontmatter_verbatim, encoding="utf-8")
|
|
|
|
# The plan's letter: ``read_provenance`` + ``trust_tier``, never ``evidence_for`` with a
|
|
# non-default key. That is not a stylistic choice — the example itself splits the two,
|
|
# putting state/reason/items_seen under ``evidence`` and the tier under ``derived``,
|
|
# because ``FalsificationEvidence.tier`` is only meaningful for the ONE key SPEC §5.3
|
|
# tiers. Branching on the primitive's own documented three-way return is test-level
|
|
# dispatch, not a second copy of production logic.
|
|
result = okf.read_provenance(path, "sources")
|
|
if concept.state == "present":
|
|
assert isinstance(result, tuple), concept.concept_id
|
|
assert len(result) == concept.items_seen, concept.concept_id
|
|
assert concept.reason is None, concept.concept_id
|
|
else:
|
|
assert isinstance(result, okf.UnreadableProvenance), concept.concept_id
|
|
assert result.reason == concept.reason, concept.concept_id
|
|
assert result.items_seen == concept.items_seen, concept.concept_id
|
|
|
|
# The verified half goes through the SHIPPED ``evidence_for`` on its DEFAULT key — the
|
|
# path where the tier derivation is the one §5.3 defines.
|
|
verified = okf.evidence_for(path)
|
|
assert okf.trust_tier(verified.entries) == concept.trust_tier, concept.concept_id
|
|
assert okf.adjudication_for(path) == concept.adjudication, concept.concept_id
|
|
seen_states.add(concept.state)
|
|
|
|
assert seen_states == {"present", "unreadable"}, (
|
|
"the example must exercise BOTH a readable and an unreadable concept -- one of each is what "
|
|
f"makes the round trip discriminating, got {sorted(seen_states)}"
|
|
)
|