"""One frontmatter SCANNER behind both readers — the seam the provenance decoder consumes. ``okf`` had two independent ``---``-delimiter loops: ``parse_frontmatter`` and ``_read_body``. Two copies of one scan is the kø-(p) shape, and here the copies had already drifted — measured, not assumed. On a file with an opening ``---`` and no closing one, ``parse_frontmatter`` consumes every remaining line as frontmatter while ``_read_body`` falls through and hands back the WHOLE file, delimiter line included. That divergence is **pinned here, not fixed.** Fixing it would move the body-rendering path both nav-goldens read, which no criterion asks for. The point of this step is that the repo gains a second *reader* of the frontmatter block and never a second *parser* of it: ``_split_frontmatter`` scans once, and each caller keeps applying its own existing rule to the result. Every expectation below was captured from the code as it stood BEFORE the split, so "unchanged" means unchanged against a measurement rather than against a recollection. """ from __future__ import annotations import ast from pathlib import Path import pytest from portfolio_optimiser import okf from portfolio_optimiser.okf import _read_body, parse_frontmatter _OKF_SOURCE = Path(okf.__file__) # The five shapes, and what today's two readers return for each. Captured 2026-09-02 by running # both functions over these exact bytes; the block case's junk ``"- { by"`` key and its # second-entry-wins value are real output, not an illustration. _SHAPES: dict[str, str] = { "flow": "---\ntype: concept\nverified: { by: human:a, at: 2026-01-01T00:00:00Z }\n---\nbody line\n", "block": ( "---\ntype: concept\nverified:\n" " - { by: human:a, at: 2026-01-01T00:00:00Z }\n" " - { by: process:b, at: 2026-01-02T00:00:00Z }\n---\nbody line\n" ), "continuation": "---\ntype: concept\ndescription: first part\n continued part\n---\nbody line\n", "none": "no frontmatter here\nsecond line\n", "unterminated": ( "---\ntype: concept\nverified: { by: human:a, at: 2026-01-01T00:00:00Z }\nbody line\n" ), } _EXPECTED_FRONTMATTER: dict[str, dict[str, str]] = { "flow": {"type": "concept", "verified": "{ by: human:a, at: 2026-01-01T00:00:00Z }"}, "block": { "type": "concept", "verified": "", "- { by": "process:b, at: 2026-01-02T00:00:00Z }", }, "continuation": {"type": "concept", "description": "first part"}, "none": {}, "unterminated": {"type": "concept", "verified": "{ by: human:a, at: 2026-01-01T00:00:00Z }"}, } _EXPECTED_BODY: dict[str, str] = { "flow": "body line\n", "block": "body line\n", "continuation": "body line\n", "none": "no frontmatter here\nsecond line\n", "unterminated": ( "---\ntype: concept\nverified: { by: human:a, at: 2026-01-01T00:00:00Z }\nbody line\n" ), } def _write(tmp_path: Path, shape: str) -> Path: path = tmp_path / f"{shape}.md" path.write_text(_SHAPES[shape], encoding="utf-8") return path @pytest.mark.parametrize("shape", sorted(_SHAPES)) def test_parse_frontmatter_is_unchanged_by_the_split(tmp_path: Path, shape: str) -> None: """``parse_frontmatter``'s dict is byte-identical to what it produced before the scanner split.""" assert parse_frontmatter(_write(tmp_path, shape)) == _EXPECTED_FRONTMATTER[shape] @pytest.mark.parametrize("shape", sorted(_SHAPES)) def test_read_body_is_unchanged_by_the_split(tmp_path: Path, shape: str) -> None: """``_read_body``'s string is byte-identical to what it produced before the scanner split.""" assert _read_body(_write(tmp_path, shape)) == _EXPECTED_BODY[shape] def test_the_two_readers_diverge_on_an_unterminated_block_and_that_is_pinned( tmp_path: Path, ) -> None: """The measured disagreement, asserted as a POSITIVE fact rather than left implicit. Without an assertion of its own, a later "tidy-up" that made the two readers agree would look like a simplification and would silently move the body-rendering path both nav-goldens read. The divergence is the reason ``_split_frontmatter`` returns ``terminated`` instead of deciding on its callers' behalf. """ path = _write(tmp_path, "unterminated") # parse_frontmatter consumed the unterminated block as if it were closed... assert parse_frontmatter(path)["type"] == "concept" # ...while _read_body treated the same file as having no frontmatter at all. assert _read_body(path) == _SHAPES["unterminated"] assert _read_body(path).startswith("---\n") def test_split_frontmatter_is_the_only_delimiter_scanner_in_okf() -> None: """The ``---`` delimiter is COMPARED against in exactly one function: ``_split_frontmatter``. ``write_concept_file`` is excluded BY NAME because it *emits* the delimiter into a formatted string — emitting is not scanning, and a whole-file substring gate could not tell the two apart. Docstrings are excluded for the same reason: prose that mentions the delimiter is not a second parser. The check walks the AST and looks only for comparisons whose right-hand side is the literal ``"---"``, which is what a scanner does and a writer never does. """ tree = ast.parse(_OKF_SOURCE.read_text(encoding="utf-8")) scanners: set[str] = set() for node in ast.walk(tree): if not isinstance(node, ast.FunctionDef): continue for inner in ast.walk(node): if isinstance(inner, ast.Compare) and any( isinstance(c, ast.Constant) and c.value == "---" for c in inner.comparators ): scanners.add(node.name) assert scanners == {"_split_frontmatter"}, ( f"the delimiter is scanned in {sorted(scanners)}; it must be scanned in exactly one place " "(write_concept_file emits it and is excluded by name)" ) def test_load_file_reads_each_file_once(tmp_path: Path) -> None: """``_load_file`` opens the document ONCE, not once per reader. Before the split it called ``parse_frontmatter`` and ``_read_body``, each of which read the file from disk — two reads of the same bytes, with the second free to see a different file than the first. """ bundle = tmp_path / "bundle" bundle.mkdir() (bundle / "a.md").write_text(_SHAPES["flow"], encoding="utf-8") reads: list[str] = [] real_read_text = Path.read_text def counting_read_text(self: Path, *args: object, **kwargs: object) -> str: reads.append(str(self)) return real_read_text(self, *args, **kwargs) # type: ignore[arg-type] with pytest.MonkeyPatch.context() as mp: mp.setattr(Path, "read_text", counting_read_text) loaded = okf._load_file(str(bundle), "a.md") assert loaded is not None assert loaded.type == "concept" assert reads.count(str(bundle / "a.md")) == 1, ( f"file was read {reads.count(str(bundle / 'a.md'))} times" ) def test_the_parsed_dict_loses_what_the_accessor_recovers(tmp_path: Path) -> None: """AMENDMENT C — the leak, shown by putting both readers on the SAME file in ONE arm. ``parse_frontmatter`` hands back ``""`` for a block-form ``verified``; the provenance accessor hands back the shape and the entry count. Asserting only the accessor's answer would leave the *reason the accessor exists* undocumented, and the reason is the whole of condition 2. ``read_provenance`` arrives in Step 4. This arm is AUTHORED here and ENABLES ITSELF the moment the symbol exists — a self-enabling skip rather than a TODO, because a note in prose is a note somebody has to remember to act on. """ read_provenance = getattr(okf, "read_provenance", None) if read_provenance is None: pytest.skip("okf.read_provenance arrives in Step 4; this arm enables itself when it does") path = _write(tmp_path, "block") assert parse_frontmatter(path)["verified"] == "" provenance = read_provenance(path, key="verified") assert provenance.entries, "the accessor recovered nothing the parser had already lost" assert len(provenance.entries) == 2 # --- Step 3: the flow-form decoder ------------------------------------------------------------ # The producer's own bytes, read from `llm-ingestion-okf` at HEAD `62b6192` (read-only). # ONE mapping: `examples/ingest-golden-okf-v0-2/expected-bundle/ingest-sales.md:9` — measured, that # is the ONLY one of 26 markdown files under `examples/` carrying a `sources:` key, so a two-mapping # concept does not exist there to read. # TWO mappings: `_render_sources`' byte-pinned output, asserted verbatim by their own # `tests/test_multi_source_provenance.py:62`. It is the authoritative referent for the N>1 form at # that HEAD, and citing it rather than hand-writing one is what keeps this arm external. _PRODUCER_ONE_SOURCE = "[{ id: golden-v0-2-sales, resource: fixture }]" _PRODUCER_TWO_SOURCES = ( "[{ id: golden-catalogue, resource: fixture }, { id: golden-db, resource: OKF_GOLDEN_SQL_DB }]" ) def test_the_producers_single_entry_bytes_decode_by_value() -> None: """S2 — the transcription arm, against the producer's real emitted line.""" assert okf.decode_flow_value(_PRODUCER_ONE_SOURCE) == ( {"id": "golden-v0-2-sales", "resource": "fixture"}, ) def test_two_mappings_decode_to_two_INTACT_entries() -> None: """AMENDMENT C — multiple sources must be READABLE, not merely refused without silence. Asserted on BOTH dicts by value. ``len() == 2`` alone stays green against a decoder that returns the first entry twice or the last one twice, which is the very last-write-wins shape this work exists to remove. """ assert okf.decode_flow_value(_PRODUCER_TWO_SOURCES) == ( {"id": "golden-catalogue", "resource": "fixture"}, {"id": "golden-db", "resource": "OKF_GOLDEN_SQL_DB"}, ) def test_entries_decode_key_agnostically() -> None: """Condition 2a — the segmented shape adds keys, and the decoder must not filter them. ``set(entry)`` is asserted against all four names: a ``len(result) == 1`` assert alone stays green against a decoder that silently drops the keys it does not recognise, which would refuse the very bundles this seam is built for. """ raw = "[{ id: s-1, resource: doc.pdf, segment_id: 4, source_offset: 128 }]" (entry,) = okf.decode_flow_value(raw) assert set(entry) == {"id", "resource", "segment_id", "source_offset"} assert entry["segment_id"] == "4" def test_a_bare_mapping_normalises_to_a_one_element_tuple() -> None: """SPEC §5.2: "Consumers MUST treat a bare mapping as a one-element list".""" assert okf.decode_flow_value("{ by: human:a, at: 2026-01-01T00:00:00Z }") == ( {"by": "human:a", "at": "2026-01-01T00:00:00Z"}, ) def test_no_yaml_1_1_coercion() -> None: """S5 — a deliberate divergence from PyYAML's resolver, asserted rather than inherited. ``yes`` resolves to the boolean ``True`` under YAML 1.1 and ``1`` to an int. Here both come back as the strings they were written as, because a value silently changing type between the file and the consumer is the coercion class this decoder refuses to import. """ (entry,) = okf.decode_flow_value("{ by: process:x, flag: yes, n: 1 }") assert entry["flag"] == "yes" assert entry["n"] == "1" assert isinstance(entry["n"], str) def test_quoted_separators_survive_the_tokeniser() -> None: """The comma and the colon-space are separators OUTSIDE quotes and ordinary text inside them. "Split on a separator" is where this class of decoder fails silently, so both separators get an arm. """ (entry,) = okf.decode_flow_value('{ resource: "a, b", note: "x: y" }') assert entry == {"resource": "a, b", "note": "x: y"} def test_an_unquoted_colon_inside_a_value_is_not_a_separator() -> None: """``by: human:jsmith@acme`` and an ISO timestamp both carry colons with no following space.""" (entry,) = okf.decode_flow_value("{ by: human:jsmith@acme, at: 2024-01-15T10:00:00Z }") assert entry == {"by": "human:jsmith@acme", "at": "2024-01-15T10:00:00Z"} @pytest.mark.parametrize( ("raw", "marker"), [ ("[ a.pdf, b.pdf ]", "bare scalar"), ("[{ id: a, resource: b }", "unterminated"), ("{ id: a, resource: b", "unterminated"), ("[{ id: a, resource: [x, y] }]", "nested"), ("{ id: a, resource: { deep: 1 } }", "nested"), ("[]", "names no source"), ('{ resource: "unclosed }', "unterminated"), ], ) def test_each_refusal_shape_raises_by_name(raw: str, marker: str) -> None: """S3 — every refusal is raised by name with its own message, never guessed past.""" with pytest.raises(okf.FlowDecodeError) as excinfo: okf.decode_flow_value(raw) assert marker in str(excinfo.value), ( f"{raw!r} refused, but not by the expected name: {excinfo.value}" ) def test_a_flow_value_continued_on_the_next_line_is_refused() -> None: """A flow form that does not fit on one line is outside the accepted subset.""" with pytest.raises(okf.FlowDecodeError) as excinfo: okf.decode_flow_value("[{ id: a,\n resource: b }]") assert "one line" in str(excinfo.value) def test_a_duplicate_key_within_one_entry_is_refused_never_last_wins() -> None: """Last-write-wins INSIDE the decoder would be the defect one level down.""" with pytest.raises(okf.FlowDecodeError) as excinfo: okf.decode_flow_value("{ by: human:a, by: process:b }") assert "duplicate" in str(excinfo.value) def test_a_verified_entry_with_no_actor_is_refused() -> None: """SPEC §5.2 makes ``by`` required within a verification event, symmetric with ``resource``. Refusing here is what lets ``trust_tier`` ASSERT that every entry names an actor instead of assuming it — the alternative, tiering an entry that names nobody, is fabricated provenance. """ with pytest.raises(okf.FlowDecodeError) as excinfo: okf.decode_flow_value("{ at: 2026-01-01T00:00:00Z }", key="verified") assert "by" in str(excinfo.value) # CONTROL: the same value under a different key is fine — the rule is `verified`-specific, # not a blanket requirement that would refuse every `sources` entry. assert okf.decode_flow_value("{ at: 2026-01-01T00:00:00Z }", key="sources") == ( {"at": "2026-01-01T00:00:00Z"}, )