portfolio-optimiser/tests/test_inert_identifier_loadbearing.py
Kjell Tore Guttormsen c66f4ae2b0
docs: general wording for the remaining example-base totals
Replace the combined concept total, the distinct-token total and the
per-level document count of earlier example bases with general wording
in prose, comments and docstrings. No constant, assertion or test data
changes.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-23 17:51:10 +02:00

295 lines
14 KiB
Python

"""P18/B1 — an identifier that stands in every document identifies none of them.
P7 made stage 0b ``item.code in grounding``: plain containment over ONE concatenated string. P16
then ran it against a delivered corpus and measured what containment cannot tell apart. The
falsification arm ``a4-indeksregulering`` proposed a 250 000 NOK saving on a single cost line whose
code was the knowledge base's OWN NAME — the catalogue designation every one of its concept
documents carries — and the whole gate said ``validated``: stage 0 was skipped (un-anchored run),
stage 0b was satisfied by the letterhead, and the checker approved.
The rule this file measures: a code grounds only if it is at least ``_GROUNDING_MIN_LENGTH``
characters AND appears in fewer than ``_GROUNDING_MAX_DOCUMENT_SHARE`` of the grounding's
DOCUMENTS — with an absolute floor, because a share over a handful of documents is not a
measurement (one of three is 33 % and says nothing).
**N and A are MEASURED, not chosen** (14.09, over the four corpora delivered at the time):
* every ``must_cite`` reference and every mandate ``affected_code`` in the four context sets: the
shortest real identifier is FOUR characters (``12.1``, ``52.1``), so ``N = 3`` sits one below the
measurement and cannot refuse anything measured;
* document frequency of every code-shaped token (``generate._IDENTIFIER_FORMS``) in each base:
well over a thousand distinct tokens and NOT ONE reaches 5 % of its base's documents. Highest anywhere 1.35 %
(6 documents); highest that a fasit names 0.67 % (3 documents); the base's own name is in every
document (100 %). ``A =
0.05`` therefore sits 3.7x above the highest real token and 20x below the defect.
**The denominator is NAMED in the refusal**, because Step 5 feeds that reason verbatim into the
next attempt's prompt: a proposer told only "ungrounded" answers with another token of the same
kind, while one told "it is in N of N documents" has been told what is wrong with it.
The arms that need a knowledge base read the package's pinned example bases (and SKIP, with the
store named, only when a user's own store lacks them); the rule's own algebra, the floor, and the
composition seam run over synthetic input and are UNCONDITIONAL.
"""
from __future__ import annotations
import json
from pathlib import Path
import pytest
from portfolio_optimiser import frozen_bundles
from portfolio_optimiser import okf
from portfolio_optimiser.ir import AffectedItem, SavingsProposal
from portfolio_optimiser.validator import (
Grounding,
Rejection,
ValidatedProposal,
_inert_in,
validate_proposal,
)
def _base(name: str) -> Path:
"""The FROZEN copy this repository pins, resolved at call time.
Absence SKIPS (a user's own store, named by ``PORTFOLIO_FROZEN_BUNDLES``, may not hold it),
drift is allowed to propagate and FAIL — a measurement of the wrong corpus is not a missing one.
"""
try:
return frozen_bundles.bundle_dir(name)
except frozen_bundles.FrozenBundleMissing as exc:
pytest.skip(str(exc))
def _grounding_over(name: str) -> Grounding:
"""The delivered base as ``run_project`` composes it: ONE document per concept file."""
bundle = okf.navigate_bundle(str(_base(name)))
return Grounding(
documents=tuple(
"\n".join([f.name, *f.frontmatter.values(), f.body]) for f in bundle.context_files
)
)
def _proposal(code: str, *, saving: float = 1000.0) -> SavingsProposal:
return SavingsProposal(
project_id="p",
measure="m",
affected_items=[AffectedItem(code=code, quantity=1.0, unit_cost=100_000.0)],
claimed_saving_nok=saving,
)
def _corpus(*, documents: int, everywhere: str, once: str) -> Grounding:
"""A synthetic grounding: one token in every document, one in exactly one."""
return Grounding(
documents=tuple(
f"{everywhere} paragraf {n}" + (f" {once}" if n == 0 else "") for n in range(documents)
)
)
# --- the measured defect --------------------------------------------------------------------
def test_the_catalogues_own_name_as_a_cost_code_is_refused() -> None:
"""(a) THE KNOWN POSITIVE. P16's paid run proposed the catalogue's own designation as a cost
code and the gate said ``validated``. That recording was made against a corpus this repository
no longer carries, so the proposal is rebuilt in the same shape against the example catalogue,
whose designation ``P900`` stands in every one of its concept documents — with
``baseline=None``, exactly the configuration under which the original said ``validated``.
CONTROL: the same base's own process number ``52.11`` still grounds (arm (b) covers every
one)."""
grounding = _grounding_over("prosesskatalog-2027")
assert _inert_in(grounding, "52.11") is None, "control: a real process number must ground"
ruling = validate_proposal(_proposal("P900"), baseline=None, grounding=grounding)
assert isinstance(ruling, Rejection)
assert "'P900'" in ruling.reason
assert "301 of the 301" in ruling.reason, (
"the refusal must name the denominator: Step 5 feeds this reason verbatim into the next "
f"attempt's prompt — got {ruling.reason!r}"
)
def test_every_fasit_reference_still_grounds() -> None:
"""(b) THE KNOWN NEGATIVE over the same corpora, with its denominator stated. A rule that made
the defect inert by making real references inert too would pass (a) perfectly."""
sets = {
"serverrom-2027": "driftskrav-2027",
"driftsavtale-2027": "prosesskatalog-2027",
}
checked = 0
for context, base in sets.items():
fasit = json.loads(Path(f"contexts/{context}/fasit.json").read_text(encoding="utf-8"))
grounding = _grounding_over(base)
for reference in sorted({c["ref"] for m in fasit["must_cite"] for c in m["concepts"]}):
assert reference in grounding.text, f"{reference!r} is absent from {base}"
assert _inert_in(grounding, reference) is None, (
f"{reference!r} is a real requirement of {base} and the rule made it inert"
)
checked += 1
assert checked == 12, f"population moved: {checked} references, expected 12"
# --- the rule's own algebra, unconditional ---------------------------------------------------
def test_a_token_in_one_document_grounds_and_one_in_all_of_them_does_not() -> None:
"""(c) The discriminator, over synthetic input so it can never be absent. Both halves in one
arm on the SAME corpus: a rule that flagged everything and one that flagged nothing each fail
exactly one of them."""
grounding = _corpus(documents=100, everywhere="KORPUS-01", once="LINJE-77-01")
assert _inert_in(grounding, "LINJE-77-01") is None
assert _inert_in(grounding, "KORPUS-01") is not None
assert isinstance(
validate_proposal(_proposal("LINJE-77-01"), grounding=grounding), ValidatedProposal
)
assert isinstance(validate_proposal(_proposal("KORPUS-01"), grounding=grounding), Rejection)
def test_a_token_too_short_to_identify_anything_is_inert() -> None:
"""(d) The length conjunct, which the SHARE does not cover: ``P900`` is four characters, so
length is not what made the measured defect inert. This is the coincidence class the
measurement did not happen to contain — a one- or two-character token is in any prose."""
grounding = Grounding(documents=("the line A is here", *("filler" for _ in range(50))))
assert _inert_in(grounding, "A") is not None
assert "too short" in str(_inert_in(grounding, "A"))
assert _inert_in(grounding, "A-1") is None, "three characters is the measured floor, not four"
def test_a_share_is_not_taken_over_a_handful_of_documents() -> None:
"""(e) The absolute floor, and the reason every pre-P18 fixture is untouched by this rule
rather than exempted from it: one document of three is 33 % and says nothing at all. Measured,
the highest ABSOLUTE document count any real identifier reaches in the four corpora is 6."""
tiny = Grounding(documents=("KODE-01 her", "KODE-01 og her", "KODE-01 og her"))
assert _inert_in(tiny, "KODE-01") is None, "3 of 3 is 100 %, and it is not a measurement"
assert isinstance(validate_proposal(_proposal("KODE-01"), grounding=tiny), ValidatedProposal)
def test_a_caller_that_declares_no_boundaries_is_byte_for_byte_the_old_gate() -> None:
"""(f) ``Grounding.of`` is the honest reading of a caller with nothing to declare, and it can
never trip the share: one document cannot reach the floor. This is what keeps every
single-document run and every pre-P18 test unchanged BY CONSTRUCTION rather than by
exemption."""
text = "en tekst som nevner KODE-99 og ellers ingenting"
single = Grounding.of(text)
assert single.text == text, "the one-document form must not reshape the text"
assert single.document_frequency("KODE-99") == 1
assert _inert_in(single, "KODE-99") is None
# --- the seam: the boundaries reach the gate from the run ------------------------------------
def test_the_run_hands_the_gate_one_document_per_concept_file() -> None:
"""(g) The COMPOSITION arm. The rule is only as good as the boundaries it is given: a run that
still composed one blob would satisfy every arm above (which builds its own ``Grounding``) and
reproduce the measured defect exactly. Driven through ``_grounding_text``, the one composer the
run passes to the gate, and asserted on the COUNT of documents rather than on the text."""
from portfolio_optimiser.generate import _grounding_text
from portfolio_optimiser.ir import CostBaseline, CostBaselineLine
from portfolio_optimiser.reference_domain import CostItem, Project
delivered = Grounding(documents=("dokument A", "dokument B", "dokument C"))
project = Project(
id="p",
name="P",
description="d",
currency="NOK",
cost_items=(
CostItem(code="PRJ-01", description="d", unit="stk", quantity=1.0, unit_cost=1.0),
),
docs_dir="/nonexistent",
)
baseline = CostBaseline(
project_id="p", items={"BAS-01": CostBaselineLine(quantity=1.0, unit_cost=1.0)}
)
composed = _grounding_text(project, baseline, delivered)
assert len(composed.documents) == 5, "each later source is ONE document, never appended to one"
assert composed.text == "\n".join(
["dokument A", "dokument B", "dokument C", "PRJ-01", "BAS-01"]
)
@pytest.mark.asyncio
async def test_the_boundaries_survive_the_real_run_and_not_only_the_composer(
tmp_path: Path,
) -> None:
"""(h) BEHAVIOURAL, over the real ``run_project`` bundle arm — and it exists because a mutation
found the gap, not because it was foreseen.
Reverting ``run.py`` to compose ONE blob instead of one document per concept file left the WHOLE
suite green (1698 passed / 5 skipped). Arm (g) above drives ``_grounding_text`` with a
``Grounding`` it builds itself, so it can never see what the RUN handed over — exactly the
vacuity this repo keeps measuring. The rule is only as good as the boundaries it is given, and
a blob has exactly one: with a single document the absolute floor can never be reached, so the
share can never fire and the measured defect returns intact.
The base is crafted so the two implementations must DISAGREE: twelve concept files all carrying
the same token, which is past ``_GROUNDING_MIN_INERT_DOCUMENTS``. Per document it is in 12 of 13
and inert; as one blob it is in 1 of 1 and grounds. The CONTROL is the same run with a code only
ONE file carries, which must still validate — otherwise the arm would also pass on a run that
rejects everything.
"""
import shutil
from portfolio_optimiser.run import run_project
from portfolio_optimiser.simulation import ScriptedChatClient
base = tmp_path / "base"
shutil.copytree(Path("shared/examples/bygg-energi-mikro"), base)
links = []
for n in range(12):
rel = f"seksjon-{n:02d}.md"
(base / rel).write_text(
f"---\ntype: concept\ntitle: Seksjon {n:02d}\n---\n\n"
"KORPUSMERKE-77 gjelder overalt i denne basen.\n"
+ ("Kostlinjen EN-ENESTE-01 star bare her.\n" if n == 0 else ""),
encoding="utf-8",
)
links.append(f"- [Seksjon {n:02d}]({rel})")
index = base / "index.md"
index.write_text(
index.read_text(encoding="utf-8") + "\n" + "\n".join(links) + "\n", encoding="utf-8"
)
async def _outcome(code: str) -> object:
reply = (
f'{{"measure":"LED-retrofit","affected_items":'
f'[{{"code":"{code}","quantity":300000,"unit_cost":1.0}}],'
f'"claimed_saving_nok":30000}}'
)
result = await run_project(
"BYGG-KONTOR-NORD",
"local",
docs_dir=str(base),
bundle_dir=str(base),
client_factory=lambda role: ScriptedChatClient(
"Reasoning holds.\nVERDICT: APPROVE" if role == "checker" else reply, role=role
),
)
return result.outcome
control = await _outcome("EN-ENESTE-01")
assert isinstance(control, ValidatedProposal), (
f"CONTROL: a code only one concept file carries must still ground — {control}"
)
everywhere = await _outcome("KORPUSMERKE-77")
assert isinstance(everywhere, Rejection), (
"a token every concept file carries grounded a proposal — the run handed the gate one blob"
)
# The NUMERATOR is the discriminator: as one blob the token is in 1 of 1, so a refusal naming
# 12 can only come from a run that kept the concept files apart.
assert "appears in 12 of the " in everywhere.reason, everywhere.reason