refactor(examples): replace sector-specific example material with generic, fictitious examples

The context sets, the packaged knowledge bases and the example bundles are
replaced by one fictitious example set about IT operations in an invented
organisation: three context sets (serverrom-2027, driftsavtale-2027 and the
two-base drift-og-avtale-2027), two synthetic knowledge bases under
src/portfolio_optimiser/data/kunnskapsbaser and two example bundles under
src/portfolio_optimiser/data/bundles. Numbers, codes and structural values in
tests and fixtures are kept; names, ids and wording change. Dated measurement
documents that only recorded runs on the replaced material are deleted.

Gate figures measured on the new set are not comparable with earlier ones.
The exclusion gate from the previous commit is green: 0 tracked files hit
outside the shared/ subtree.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Kjell Tore Guttormsen 2026-09-23 15:04:21 +02:00
commit 37547fe292
Signed by: ktg
SSH key fingerprint: SHA256:JakMjO6FTBBzN0Bhfj9saOoEjaFxlSdYuZQQpM/lF9Q
1147 changed files with 24138 additions and 9503 deletions

View file

@ -16,8 +16,8 @@ measure without sharing a word with the name someone gave it — which is precis
alternative rule P21/C1 measured and rejected failed, one rung over.
**Measured before it was built**, offline against the six traces: the rule speaks on **10 of 12**
declarations and stays quiet on 2 (both fv412, on ``materialer``). A rule that spoke on 12 of 12,
or on 0 of 12, could not tell the two classes apart.
declarations and stays quiet on 2 (both in one context set, on ``materialer``). A rule that spoke
on 12 of 12, or on 0 of 12, could not tell the two classes apart.
"""
from __future__ import annotations
@ -33,24 +33,25 @@ from portfolio_optimiser.explore import ToolCall, navigator_tools
from portfolio_optimiser.mandate import Approach, Mandate
from portfolio_optimiser.run import run_project
_EXAMPLES = Path(__file__).resolve().parents[1] / "shared" / "examples"
_TUNNEL = _EXAMPLES / "tunnel-hauglia"
_BASE_ID = "tunnel-hauglia"
_BUNDLES = Path(__file__).resolve().parents[1] / "src" / "portfolio_optimiser" / "data" / "bundles"
_BASE = _BUNDLES / "driftssenter-kjoling"
_BASE_ID = "driftssenter-kjoling"
#: The fixture document the arms declare. Its title is ASCII-clean, which is what lets a label
#: share a word with it without a marker carrying a multibyte character into a scripted run.
_DOC = "tiltak-portalskjerming.md"
#: The fixture document the arms declare. The words the arms match on are ASCII-clean, which is
#: what lets a label share a word with it without a marker carrying a multibyte character into a
#: scripted run.
_DOC = "tiltak-kaldgangsinnkapsling.md"
#: A direction whose words are IN that title ("Portalskjerming: senke L20 ...").
_MATCHING = "Billigere portalskjerming"
#: A direction whose words are IN that title ("Kaldgangsinnkapsling: senke varmetilskuddet ...").
_MATCHING = "Billigere kaldgangsinnkapsling"
#: A direction that shares nothing with it. Checked against the document's own tokens, both ways.
_FOREIGN = "Asfaltdekke gjenbruk"
_FOREIGN = "Lagringsenhet gjenbruk"
def _wired(labels: tuple[str, ...]) -> tuple[dict[str, Any], list[ToolCall], list[Any]]:
opened: list[ToolCall] = []
declared: list[Any] = []
tools = navigator_tools((str(_TUNNEL),), opened=opened, requirements=declared, labels=labels)
tools = navigator_tools((str(_BASE),), opened=opened, requirements=declared, labels=labels)
return {t.name: t for t in tools}, opened, declared
@ -58,7 +59,7 @@ def _declare(
labels: tuple[str, ...], *, ref: str = "Krav 1.1-1"
) -> tuple[dict[str, Any], list[Any]]:
tools, opened, declared = _wired(labels)
for name in [f.name for f in okf.navigate_bundle(str(_TUNNEL)).context_files][:3]:
for name in [f.name for f in okf.navigate_bundle(str(_BASE)).context_files][:3]:
opened.append(ToolCall(name="read_file", bundle_id=_BASE_ID, path=name))
opened.append(ToolCall(name="read_file", bundle_id=_BASE_ID, path=_DOC))
answer = tools["declare_requirement"].func(
@ -74,8 +75,8 @@ def test_a_direction_that_shares_a_word_is_told_which_one() -> None:
"""KNOWN-POSITIVE. The document's title carries the direction's own word, and the reply says
so — the half that keeps the report from being one that only ever complains."""
answer, _ = _declare((_MATCHING,))
assert answer["overlap"] == ["portalskjerming"], answer["overlap"]
assert "portalskjerming" in answer["compare"]
assert answer["overlap"] == ["kaldgangsinnkapsling"], answer["overlap"]
assert "kaldgangsinnkapsling" in answer["compare"]
def test_a_direction_that_shares_nothing_is_told_that_too() -> None:
@ -108,7 +109,7 @@ def test_the_comparison_never_reads_the_callers_own_ref() -> None:
input can only ever agree — P20/A1's rule (read off the base, never off the arguments)
applied to the half P20 did not reach. A ``ref`` stuffed with the direction's words must not
manufacture an overlap."""
answer, _ = _declare((_FOREIGN,), ref="Asfaltdekke gjenbruk krav")
answer, _ = _declare((_FOREIGN,), ref="Lagringsenhet gjenbruk krav")
assert answer["overlap"] == [], answer["overlap"]
@ -139,11 +140,13 @@ def test_without_directions_the_reply_is_the_one_p20_shipped() -> None:
def test_matching_is_generous_in_both_directions() -> None:
"""The failure direction chosen on purpose. A direction whose word is a PREFIX of the
document's own word counts, and so does the reverse — Norwegian inflects ("rundkjoring" vs
"Rundkjoringer") and a strict rule would report "no overlap" on a declaration that was right,
document's own word counts, and so does the reverse — Norwegian inflects ("sikkerhetskopi" vs
"Sikkerhetskopier") and a strict rule would report "no overlap" on a declaration that was right,
which is the only one of the two errors that can push a model away from a correct answer."""
# label word LONGER than the document's own ("portalskjerming" is a prefix of it)
assert _declare(("Portalskjermingen paa nordsiden",))[0]["overlap"] == ["portalskjermingen"]
# label word LONGER than the document's own ("kaldgangsinnkapsling" is a prefix of it)
assert _declare(("Kaldgangsinnkapslingen paa nordsiden",))[0]["overlap"] == [
"kaldgangsinnkapslingen"
]
# and SHORTER: the document says "senke", the direction "senkekostnader"
assert _declare(("Senkekostnader",))[0]["overlap"] == ["senkekostnader"]
@ -158,7 +161,7 @@ def test_short_words_cannot_manufacture_an_overlap() -> None:
# --- (h) the commission's labels actually REACH the tool in a run --------------------------------
_PID = "TUNNEL-HAUGLIA"
_PID = "DRIFTSSENTER-KJOLING"
_VERDICT_INPUT = {"decision": "approved", "rationale": "expert reviewed (test)"}
_VALID_REPLY = (
'{"measure":"LED-retrofit","affected_items":'
@ -174,7 +177,7 @@ async def test_a_commissioned_run_reaches_the_tool_with_its_own_directions() ->
a lint; this drives the real debate with a step manuscript that declares, and reads the
comparison back out of the tool's own answer. Without ``labels=`` in ``run.py`` the reply
carries no ``compare`` at all and this arm falls."""
concepts = [f.name for f in okf.navigate_bundle(str(_TUNNEL)).context_files]
concepts = [f.name for f in okf.navigate_bundle(str(_BASE)).context_files]
script = {
"proposer": [
*(
@ -218,8 +221,8 @@ async def test_a_commissioned_run_reaches_the_tool_with_its_own_directions() ->
await run_project(
_PID,
"local",
docs_dir=str(_TUNNEL),
bundle_dir=str(_TUNNEL),
docs_dir=str(_BASE),
bundle_dir=str(_BASE),
verdict_input=_VERDICT_INPUT,
client_factory=factory,
mandate=Mandate(