317 lines
14 KiB
Python
317 lines
14 KiB
Python
"""One frontmatter SCANNER behind both readers — the seam the provenance decoder consumes.
|
|
|
|
``okf`` had two independent ``---``-delimiter loops: ``parse_frontmatter`` and ``_read_body``.
|
|
Two copies of one scan is the kø-(p) shape, and here the copies had already drifted — measured,
|
|
not assumed. On a file with an opening ``---`` and no closing one, ``parse_frontmatter`` consumes
|
|
every remaining line as frontmatter while ``_read_body`` falls through and hands back the WHOLE
|
|
file, delimiter line included.
|
|
|
|
That divergence is **pinned here, not fixed.** Fixing it would move the body-rendering path both
|
|
nav-goldens read, which no criterion asks for. The point of this step is that the repo gains a
|
|
second *reader* of the frontmatter block and never a second *parser* of it: ``_split_frontmatter``
|
|
scans once, and each caller keeps applying its own existing rule to the result.
|
|
|
|
Every expectation below was captured from the code as it stood BEFORE the split, so "unchanged"
|
|
means unchanged against a measurement rather than against a recollection.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import ast
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from portfolio_optimiser import okf
|
|
from portfolio_optimiser.okf import _read_body, parse_frontmatter
|
|
|
|
_OKF_SOURCE = Path(okf.__file__)
|
|
|
|
# The five shapes, and what today's two readers return for each. Captured 2026-09-02 by running
|
|
# both functions over these exact bytes; the block case's junk ``"- { by"`` key and its
|
|
# second-entry-wins value are real output, not an illustration.
|
|
_SHAPES: dict[str, str] = {
|
|
"flow": "---\ntype: concept\nverified: { by: human:a, at: 2026-01-01T00:00:00Z }\n---\nbody line\n",
|
|
"block": (
|
|
"---\ntype: concept\nverified:\n"
|
|
" - { by: human:a, at: 2026-01-01T00:00:00Z }\n"
|
|
" - { by: process:b, at: 2026-01-02T00:00:00Z }\n---\nbody line\n"
|
|
),
|
|
"continuation": "---\ntype: concept\ndescription: first part\n continued part\n---\nbody line\n",
|
|
"none": "no frontmatter here\nsecond line\n",
|
|
"unterminated": (
|
|
"---\ntype: concept\nverified: { by: human:a, at: 2026-01-01T00:00:00Z }\nbody line\n"
|
|
),
|
|
}
|
|
|
|
_EXPECTED_FRONTMATTER: dict[str, dict[str, str]] = {
|
|
"flow": {"type": "concept", "verified": "{ by: human:a, at: 2026-01-01T00:00:00Z }"},
|
|
"block": {
|
|
"type": "concept",
|
|
"verified": "",
|
|
"- { by": "process:b, at: 2026-01-02T00:00:00Z }",
|
|
},
|
|
"continuation": {"type": "concept", "description": "first part"},
|
|
"none": {},
|
|
"unterminated": {"type": "concept", "verified": "{ by: human:a, at: 2026-01-01T00:00:00Z }"},
|
|
}
|
|
|
|
_EXPECTED_BODY: dict[str, str] = {
|
|
"flow": "body line\n",
|
|
"block": "body line\n",
|
|
"continuation": "body line\n",
|
|
"none": "no frontmatter here\nsecond line\n",
|
|
"unterminated": (
|
|
"---\ntype: concept\nverified: { by: human:a, at: 2026-01-01T00:00:00Z }\nbody line\n"
|
|
),
|
|
}
|
|
|
|
|
|
def _write(tmp_path: Path, shape: str) -> Path:
|
|
path = tmp_path / f"{shape}.md"
|
|
path.write_text(_SHAPES[shape], encoding="utf-8")
|
|
return path
|
|
|
|
|
|
@pytest.mark.parametrize("shape", sorted(_SHAPES))
|
|
def test_parse_frontmatter_is_unchanged_by_the_split(tmp_path: Path, shape: str) -> None:
|
|
"""``parse_frontmatter``'s dict is byte-identical to what it produced before the scanner split."""
|
|
assert parse_frontmatter(_write(tmp_path, shape)) == _EXPECTED_FRONTMATTER[shape]
|
|
|
|
|
|
@pytest.mark.parametrize("shape", sorted(_SHAPES))
|
|
def test_read_body_is_unchanged_by_the_split(tmp_path: Path, shape: str) -> None:
|
|
"""``_read_body``'s string is byte-identical to what it produced before the scanner split."""
|
|
assert _read_body(_write(tmp_path, shape)) == _EXPECTED_BODY[shape]
|
|
|
|
|
|
def test_the_two_readers_diverge_on_an_unterminated_block_and_that_is_pinned(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
"""The measured disagreement, asserted as a POSITIVE fact rather than left implicit.
|
|
|
|
Without an assertion of its own, a later "tidy-up" that made the two readers agree would look
|
|
like a simplification and would silently move the body-rendering path both nav-goldens read.
|
|
The divergence is the reason ``_split_frontmatter`` returns ``terminated`` instead of deciding
|
|
on its callers' behalf.
|
|
"""
|
|
path = _write(tmp_path, "unterminated")
|
|
|
|
# parse_frontmatter consumed the unterminated block as if it were closed...
|
|
assert parse_frontmatter(path)["type"] == "concept"
|
|
# ...while _read_body treated the same file as having no frontmatter at all.
|
|
assert _read_body(path) == _SHAPES["unterminated"]
|
|
assert _read_body(path).startswith("---\n")
|
|
|
|
|
|
def test_split_frontmatter_is_the_only_delimiter_scanner_in_okf() -> None:
|
|
"""The ``---`` delimiter is COMPARED against in exactly one function: ``_split_frontmatter``.
|
|
|
|
``write_concept_file`` is excluded BY NAME because it *emits* the delimiter into a formatted
|
|
string — emitting is not scanning, and a whole-file substring gate could not tell the two
|
|
apart. Docstrings are excluded for the same reason: prose that mentions the delimiter is not
|
|
a second parser. The check walks the AST and looks only for comparisons whose right-hand side
|
|
is the literal ``"---"``, which is what a scanner does and a writer never does.
|
|
"""
|
|
tree = ast.parse(_OKF_SOURCE.read_text(encoding="utf-8"))
|
|
scanners: set[str] = set()
|
|
for node in ast.walk(tree):
|
|
if not isinstance(node, ast.FunctionDef):
|
|
continue
|
|
for inner in ast.walk(node):
|
|
if isinstance(inner, ast.Compare) and any(
|
|
isinstance(c, ast.Constant) and c.value == "---" for c in inner.comparators
|
|
):
|
|
scanners.add(node.name)
|
|
assert scanners == {"_split_frontmatter"}, (
|
|
f"the delimiter is scanned in {sorted(scanners)}; it must be scanned in exactly one place "
|
|
"(write_concept_file emits it and is excluded by name)"
|
|
)
|
|
|
|
|
|
def test_load_file_reads_each_file_once(tmp_path: Path) -> None:
|
|
"""``_load_file`` opens the document ONCE, not once per reader.
|
|
|
|
Before the split it called ``parse_frontmatter`` and ``_read_body``, each of which read the
|
|
file from disk — two reads of the same bytes, with the second free to see a different file
|
|
than the first.
|
|
"""
|
|
bundle = tmp_path / "bundle"
|
|
bundle.mkdir()
|
|
(bundle / "a.md").write_text(_SHAPES["flow"], encoding="utf-8")
|
|
|
|
reads: list[str] = []
|
|
real_read_text = Path.read_text
|
|
|
|
def counting_read_text(self: Path, *args: object, **kwargs: object) -> str:
|
|
reads.append(str(self))
|
|
return real_read_text(self, *args, **kwargs) # type: ignore[arg-type]
|
|
|
|
with pytest.MonkeyPatch.context() as mp:
|
|
mp.setattr(Path, "read_text", counting_read_text)
|
|
loaded = okf._load_file(str(bundle), "a.md")
|
|
|
|
assert loaded is not None
|
|
assert loaded.type == "concept"
|
|
assert reads.count(str(bundle / "a.md")) == 1, (
|
|
f"file was read {reads.count(str(bundle / 'a.md'))} times"
|
|
)
|
|
|
|
|
|
def test_the_parsed_dict_loses_what_the_accessor_recovers(tmp_path: Path) -> None:
|
|
"""AMENDMENT C — the leak, shown by putting both readers on the SAME file in ONE arm.
|
|
|
|
``parse_frontmatter`` hands back ``""`` for a block-form ``verified``; the provenance accessor
|
|
hands back the shape and the entry count. Asserting only the accessor's answer would leave the
|
|
*reason the accessor exists* undocumented, and the reason is the whole of condition 2.
|
|
|
|
``read_provenance`` arrives in Step 4. This arm is AUTHORED here and ENABLES ITSELF the moment
|
|
the symbol exists — a self-enabling skip rather than a TODO, because a note in prose is a note
|
|
somebody has to remember to act on.
|
|
"""
|
|
read_provenance = getattr(okf, "read_provenance", None)
|
|
if read_provenance is None:
|
|
pytest.skip("okf.read_provenance arrives in Step 4; this arm enables itself when it does")
|
|
|
|
path = _write(tmp_path, "block")
|
|
assert parse_frontmatter(path)["verified"] == ""
|
|
|
|
provenance = read_provenance(path, key="verified")
|
|
assert provenance.entries, "the accessor recovered nothing the parser had already lost"
|
|
assert len(provenance.entries) == 2
|
|
|
|
|
|
# --- Step 3: the flow-form decoder ------------------------------------------------------------
|
|
|
|
# The producer's own bytes, read from `llm-ingestion-okf` at HEAD `62b6192` (read-only).
|
|
# ONE mapping: `examples/ingest-golden-okf-v0-2/expected-bundle/ingest-sales.md:9` — measured, that
|
|
# is the ONLY one of 26 markdown files under `examples/` carrying a `sources:` key, so a two-mapping
|
|
# concept does not exist there to read.
|
|
# TWO mappings: `_render_sources`' byte-pinned output, asserted verbatim by their own
|
|
# `tests/test_multi_source_provenance.py:62`. It is the authoritative referent for the N>1 form at
|
|
# that HEAD, and citing it rather than hand-writing one is what keeps this arm external.
|
|
_PRODUCER_ONE_SOURCE = "[{ id: golden-v0-2-sales, resource: fixture }]"
|
|
_PRODUCER_TWO_SOURCES = (
|
|
"[{ id: golden-catalogue, resource: fixture }, { id: golden-db, resource: OKF_GOLDEN_SQL_DB }]"
|
|
)
|
|
|
|
|
|
def test_the_producers_single_entry_bytes_decode_by_value() -> None:
|
|
"""S2 — the transcription arm, against the producer's real emitted line."""
|
|
assert okf.decode_flow_value(_PRODUCER_ONE_SOURCE) == (
|
|
{"id": "golden-v0-2-sales", "resource": "fixture"},
|
|
)
|
|
|
|
|
|
def test_two_mappings_decode_to_two_INTACT_entries() -> None:
|
|
"""AMENDMENT C — multiple sources must be READABLE, not merely refused without silence.
|
|
|
|
Asserted on BOTH dicts by value. ``len() == 2`` alone stays green against a decoder that
|
|
returns the first entry twice or the last one twice, which is the very last-write-wins shape
|
|
this work exists to remove.
|
|
"""
|
|
assert okf.decode_flow_value(_PRODUCER_TWO_SOURCES) == (
|
|
{"id": "golden-catalogue", "resource": "fixture"},
|
|
{"id": "golden-db", "resource": "OKF_GOLDEN_SQL_DB"},
|
|
)
|
|
|
|
|
|
def test_entries_decode_key_agnostically() -> None:
|
|
"""Condition 2a — the segmented shape adds keys, and the decoder must not filter them.
|
|
|
|
``set(entry)`` is asserted against all four names: a ``len(result) == 1`` assert alone stays
|
|
green against a decoder that silently drops the keys it does not recognise, which would refuse
|
|
the very bundles this seam is built for.
|
|
"""
|
|
raw = "[{ id: s-1, resource: doc.pdf, segment_id: 4, source_offset: 128 }]"
|
|
(entry,) = okf.decode_flow_value(raw)
|
|
assert set(entry) == {"id", "resource", "segment_id", "source_offset"}
|
|
assert entry["segment_id"] == "4"
|
|
|
|
|
|
def test_a_bare_mapping_normalises_to_a_one_element_tuple() -> None:
|
|
"""SPEC §5.2: "Consumers MUST treat a bare mapping as a one-element list"."""
|
|
assert okf.decode_flow_value("{ by: human:a, at: 2026-01-01T00:00:00Z }") == (
|
|
{"by": "human:a", "at": "2026-01-01T00:00:00Z"},
|
|
)
|
|
|
|
|
|
def test_no_yaml_1_1_coercion() -> None:
|
|
"""S5 — a deliberate divergence from PyYAML's resolver, asserted rather than inherited.
|
|
|
|
``yes`` resolves to the boolean ``True`` under YAML 1.1 and ``1`` to an int. Here both come
|
|
back as the strings they were written as, because a value silently changing type between the
|
|
file and the consumer is the coercion class this decoder refuses to import.
|
|
"""
|
|
(entry,) = okf.decode_flow_value("{ by: process:x, flag: yes, n: 1 }")
|
|
assert entry["flag"] == "yes"
|
|
assert entry["n"] == "1"
|
|
assert isinstance(entry["n"], str)
|
|
|
|
|
|
def test_quoted_separators_survive_the_tokeniser() -> None:
|
|
"""The comma and the colon-space are separators OUTSIDE quotes and ordinary text inside them.
|
|
|
|
"Split on a separator" is where this class of decoder fails silently, so both separators get
|
|
an arm.
|
|
"""
|
|
(entry,) = okf.decode_flow_value('{ resource: "a, b", note: "x: y" }')
|
|
assert entry == {"resource": "a, b", "note": "x: y"}
|
|
|
|
|
|
def test_an_unquoted_colon_inside_a_value_is_not_a_separator() -> None:
|
|
"""``by: human:jsmith@acme`` and an ISO timestamp both carry colons with no following space."""
|
|
(entry,) = okf.decode_flow_value("{ by: human:jsmith@acme, at: 2024-01-15T10:00:00Z }")
|
|
assert entry == {"by": "human:jsmith@acme", "at": "2024-01-15T10:00:00Z"}
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("raw", "marker"),
|
|
[
|
|
("[ a.pdf, b.pdf ]", "bare scalar"),
|
|
("[{ id: a, resource: b }", "unterminated"),
|
|
("{ id: a, resource: b", "unterminated"),
|
|
("[{ id: a, resource: [x, y] }]", "nested"),
|
|
("{ id: a, resource: { deep: 1 } }", "nested"),
|
|
("[]", "names no source"),
|
|
('{ resource: "unclosed }', "unterminated"),
|
|
],
|
|
)
|
|
def test_each_refusal_shape_raises_by_name(raw: str, marker: str) -> None:
|
|
"""S3 — every refusal is raised by name with its own message, never guessed past."""
|
|
with pytest.raises(okf.FlowDecodeError) as excinfo:
|
|
okf.decode_flow_value(raw)
|
|
assert marker in str(excinfo.value), (
|
|
f"{raw!r} refused, but not by the expected name: {excinfo.value}"
|
|
)
|
|
|
|
|
|
def test_a_flow_value_continued_on_the_next_line_is_refused() -> None:
|
|
"""A flow form that does not fit on one line is outside the accepted subset."""
|
|
with pytest.raises(okf.FlowDecodeError) as excinfo:
|
|
okf.decode_flow_value("[{ id: a,\n resource: b }]")
|
|
assert "one line" in str(excinfo.value)
|
|
|
|
|
|
def test_a_duplicate_key_within_one_entry_is_refused_never_last_wins() -> None:
|
|
"""Last-write-wins INSIDE the decoder would be the defect one level down."""
|
|
with pytest.raises(okf.FlowDecodeError) as excinfo:
|
|
okf.decode_flow_value("{ by: human:a, by: process:b }")
|
|
assert "duplicate" in str(excinfo.value)
|
|
|
|
|
|
def test_a_verified_entry_with_no_actor_is_refused() -> None:
|
|
"""SPEC §5.2 makes ``by`` required within a verification event, symmetric with ``resource``.
|
|
|
|
Refusing here is what lets ``trust_tier`` ASSERT that every entry names an actor instead of
|
|
assuming it — the alternative, tiering an entry that names nobody, is fabricated provenance.
|
|
"""
|
|
with pytest.raises(okf.FlowDecodeError) as excinfo:
|
|
okf.decode_flow_value("{ at: 2026-01-01T00:00:00Z }", key="verified")
|
|
assert "by" in str(excinfo.value)
|
|
|
|
# CONTROL: the same value under a different key is fine — the rule is `verified`-specific,
|
|
# not a blanket requirement that would refuse every `sources` entry.
|
|
assert okf.decode_flow_value("{ at: 2026-01-01T00:00:00Z }", key="sources") == (
|
|
{"at": "2026-01-01T00:00:00Z"},
|
|
)
|