refactor(okf): one frontmatter scanner behind both readers, contracts unchanged
Co-Authored-By: Claude <claude-opus-5>
This commit is contained in:
parent
809fa04406
commit
ab2c509337
2 changed files with 242 additions and 19 deletions
181
tests/test_provenance_decoder_loadbearing.py
Normal file
181
tests/test_provenance_decoder_loadbearing.py
Normal file
|
|
@ -0,0 +1,181 @@
|
|||
"""One frontmatter SCANNER behind both readers — the seam the provenance decoder consumes.
|
||||
|
||||
``okf`` had two independent ``---``-delimiter loops: ``parse_frontmatter`` and ``_read_body``.
|
||||
Two copies of one scan is the kø-(p) shape, and here the copies had already drifted — measured,
|
||||
not assumed. On a file with an opening ``---`` and no closing one, ``parse_frontmatter`` consumes
|
||||
every remaining line as frontmatter while ``_read_body`` falls through and hands back the WHOLE
|
||||
file, delimiter line included.
|
||||
|
||||
That divergence is **pinned here, not fixed.** Fixing it would move the body-rendering path both
|
||||
nav-goldens read, which no criterion asks for. The point of this step is that the repo gains a
|
||||
second *reader* of the frontmatter block and never a second *parser* of it: ``_split_frontmatter``
|
||||
scans once, and each caller keeps applying its own existing rule to the result.
|
||||
|
||||
Every expectation below was captured from the code as it stood BEFORE the split, so "unchanged"
|
||||
means unchanged against a measurement rather than against a recollection.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from portfolio_optimiser import okf
|
||||
from portfolio_optimiser.okf import _read_body, parse_frontmatter
|
||||
|
||||
_OKF_SOURCE = Path(okf.__file__)
|
||||
|
||||
# The five shapes, and what today's two readers return for each. Captured 2026-09-02 by running
|
||||
# both functions over these exact bytes; the block case's junk ``"- { by"`` key and its
|
||||
# second-entry-wins value are real output, not an illustration.
|
||||
_SHAPES: dict[str, str] = {
|
||||
"flow": "---\ntype: concept\nverified: { by: human:a, at: 2026-01-01T00:00:00Z }\n---\nbody line\n",
|
||||
"block": (
|
||||
"---\ntype: concept\nverified:\n"
|
||||
" - { by: human:a, at: 2026-01-01T00:00:00Z }\n"
|
||||
" - { by: process:b, at: 2026-01-02T00:00:00Z }\n---\nbody line\n"
|
||||
),
|
||||
"continuation": "---\ntype: concept\ndescription: first part\n continued part\n---\nbody line\n",
|
||||
"none": "no frontmatter here\nsecond line\n",
|
||||
"unterminated": (
|
||||
"---\ntype: concept\nverified: { by: human:a, at: 2026-01-01T00:00:00Z }\nbody line\n"
|
||||
),
|
||||
}
|
||||
|
||||
_EXPECTED_FRONTMATTER: dict[str, dict[str, str]] = {
|
||||
"flow": {"type": "concept", "verified": "{ by: human:a, at: 2026-01-01T00:00:00Z }"},
|
||||
"block": {
|
||||
"type": "concept",
|
||||
"verified": "",
|
||||
"- { by": "process:b, at: 2026-01-02T00:00:00Z }",
|
||||
},
|
||||
"continuation": {"type": "concept", "description": "first part"},
|
||||
"none": {},
|
||||
"unterminated": {"type": "concept", "verified": "{ by: human:a, at: 2026-01-01T00:00:00Z }"},
|
||||
}
|
||||
|
||||
_EXPECTED_BODY: dict[str, str] = {
|
||||
"flow": "body line\n",
|
||||
"block": "body line\n",
|
||||
"continuation": "body line\n",
|
||||
"none": "no frontmatter here\nsecond line\n",
|
||||
"unterminated": (
|
||||
"---\ntype: concept\nverified: { by: human:a, at: 2026-01-01T00:00:00Z }\nbody line\n"
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def _write(tmp_path: Path, shape: str) -> Path:
|
||||
path = tmp_path / f"{shape}.md"
|
||||
path.write_text(_SHAPES[shape], encoding="utf-8")
|
||||
return path
|
||||
|
||||
|
||||
@pytest.mark.parametrize("shape", sorted(_SHAPES))
|
||||
def test_parse_frontmatter_is_unchanged_by_the_split(tmp_path: Path, shape: str) -> None:
|
||||
"""``parse_frontmatter``'s dict is byte-identical to what it produced before the scanner split."""
|
||||
assert parse_frontmatter(_write(tmp_path, shape)) == _EXPECTED_FRONTMATTER[shape]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("shape", sorted(_SHAPES))
|
||||
def test_read_body_is_unchanged_by_the_split(tmp_path: Path, shape: str) -> None:
|
||||
"""``_read_body``'s string is byte-identical to what it produced before the scanner split."""
|
||||
assert _read_body(_write(tmp_path, shape)) == _EXPECTED_BODY[shape]
|
||||
|
||||
|
||||
def test_the_two_readers_diverge_on_an_unterminated_block_and_that_is_pinned(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
"""The measured disagreement, asserted as a POSITIVE fact rather than left implicit.
|
||||
|
||||
Without an assertion of its own, a later "tidy-up" that made the two readers agree would look
|
||||
like a simplification and would silently move the body-rendering path both nav-goldens read.
|
||||
The divergence is the reason ``_split_frontmatter`` returns ``terminated`` instead of deciding
|
||||
on its callers' behalf.
|
||||
"""
|
||||
path = _write(tmp_path, "unterminated")
|
||||
|
||||
# parse_frontmatter consumed the unterminated block as if it were closed...
|
||||
assert parse_frontmatter(path)["type"] == "concept"
|
||||
# ...while _read_body treated the same file as having no frontmatter at all.
|
||||
assert _read_body(path) == _SHAPES["unterminated"]
|
||||
assert _read_body(path).startswith("---\n")
|
||||
|
||||
|
||||
def test_split_frontmatter_is_the_only_delimiter_scanner_in_okf() -> None:
|
||||
"""The ``---`` delimiter is COMPARED against in exactly one function: ``_split_frontmatter``.
|
||||
|
||||
``write_concept_file`` is excluded BY NAME because it *emits* the delimiter into a formatted
|
||||
string — emitting is not scanning, and a whole-file substring gate could not tell the two
|
||||
apart. Docstrings are excluded for the same reason: prose that mentions the delimiter is not
|
||||
a second parser. The check walks the AST and looks only for comparisons whose right-hand side
|
||||
is the literal ``"---"``, which is what a scanner does and a writer never does.
|
||||
"""
|
||||
tree = ast.parse(_OKF_SOURCE.read_text(encoding="utf-8"))
|
||||
scanners: set[str] = set()
|
||||
for node in ast.walk(tree):
|
||||
if not isinstance(node, ast.FunctionDef):
|
||||
continue
|
||||
for inner in ast.walk(node):
|
||||
if isinstance(inner, ast.Compare) and any(
|
||||
isinstance(c, ast.Constant) and c.value == "---" for c in inner.comparators
|
||||
):
|
||||
scanners.add(node.name)
|
||||
assert scanners == {"_split_frontmatter"}, (
|
||||
f"the delimiter is scanned in {sorted(scanners)}; it must be scanned in exactly one place "
|
||||
"(write_concept_file emits it and is excluded by name)"
|
||||
)
|
||||
|
||||
|
||||
def test_load_file_reads_each_file_once(tmp_path: Path) -> None:
|
||||
"""``_load_file`` opens the document ONCE, not once per reader.
|
||||
|
||||
Before the split it called ``parse_frontmatter`` and ``_read_body``, each of which read the
|
||||
file from disk — two reads of the same bytes, with the second free to see a different file
|
||||
than the first.
|
||||
"""
|
||||
bundle = tmp_path / "bundle"
|
||||
bundle.mkdir()
|
||||
(bundle / "a.md").write_text(_SHAPES["flow"], encoding="utf-8")
|
||||
|
||||
reads: list[str] = []
|
||||
real_read_text = Path.read_text
|
||||
|
||||
def counting_read_text(self: Path, *args: object, **kwargs: object) -> str:
|
||||
reads.append(str(self))
|
||||
return real_read_text(self, *args, **kwargs) # type: ignore[arg-type]
|
||||
|
||||
with pytest.MonkeyPatch.context() as mp:
|
||||
mp.setattr(Path, "read_text", counting_read_text)
|
||||
loaded = okf._load_file(str(bundle), "a.md")
|
||||
|
||||
assert loaded is not None
|
||||
assert loaded.type == "concept"
|
||||
assert reads.count(str(bundle / "a.md")) == 1, (
|
||||
f"file was read {reads.count(str(bundle / 'a.md'))} times"
|
||||
)
|
||||
|
||||
|
||||
def test_the_parsed_dict_loses_what_the_accessor_recovers(tmp_path: Path) -> None:
|
||||
"""AMENDMENT C — the leak, shown by putting both readers on the SAME file in ONE arm.
|
||||
|
||||
``parse_frontmatter`` hands back ``""`` for a block-form ``verified``; the provenance accessor
|
||||
hands back the shape and the entry count. Asserting only the accessor's answer would leave the
|
||||
*reason the accessor exists* undocumented, and the reason is the whole of condition 2.
|
||||
|
||||
``read_provenance`` arrives in Step 4. This arm is AUTHORED here and ENABLES ITSELF the moment
|
||||
the symbol exists — a self-enabling skip rather than a TODO, because a note in prose is a note
|
||||
somebody has to remember to act on.
|
||||
"""
|
||||
read_provenance = getattr(okf, "read_provenance", None)
|
||||
if read_provenance is None:
|
||||
pytest.skip("okf.read_provenance arrives in Step 4; this arm enables itself when it does")
|
||||
|
||||
path = _write(tmp_path, "block")
|
||||
assert parse_frontmatter(path)["verified"] == ""
|
||||
|
||||
provenance = read_provenance(path, key="verified")
|
||||
assert provenance.entries, "the accessor recovered nothing the parser had already lost"
|
||||
assert len(provenance.entries) == 2
|
||||
Loading…
Add table
Add a link
Reference in a new issue