"""SPEC 5.1 `sources` with more than one entry, and the form it is emitted in. PM decision B6 asked for a list-taking `_render_sources` so a concept can record more than one source. It also prescribed the BLOCK list as the emitted form. The list is delivered; the block form is not, and this file carries the measurement rather than the argument. Three facts, each pinned by a test below: - **Our own parser loses a block list entirely.** `parse_frontmatter` is line-oriented and skips indented lines, so `sources:` followed by ` - id: a` round-trips to an EMPTY value with every entry gone -- silently. A provenance record we cannot read back is worse than one we never wrote. - **The consumer's decoder reads the flow sequence and refuses the block one.** `portfolio-optimiser`'s `decode_flow_value` accepts `[{ k: v }, { k: v }]` -- plural -- and classifies a block sequence as `UnreadableProvenance`. Emitting block would hand the consumer that asked for multi-source exactly the state it reports as unreadable. - **The order's own acceptance test settles it.** It asks for a round trip through our `parse_frontmatter` equivalent. No block form can pass that. So the goal is delivered in the form that reaches a reader: one flow sequence of N flow mappings, on one line. A single source stays byte-identical, which is what keeps all six goldens unmoved. """ from __future__ import annotations from pathlib import Path import pytest from llm_ingestion_okf.errors import MaterializationError from llm_ingestion_okf.manifest import FileSource, HttpSource, Source, SqlSource from llm_ingestion_okf.materialize import _render_sources, parse_frontmatter CATALOGUE = FileSource(id="golden-catalogue", root="fixture") DATABASE = SqlSource(id="golden-db", connection_ref="OKF_GOLDEN_SQL_DB") API = HttpSource(id="golden-api", base_url="https://golden.example.test") def _document(sources: str) -> str: return ( "---\n" "type: dataset\n" "title: Orders\n" "generated: { by: process:okf-ingest, at: 2026-01-01T00:00:00Z }\n" f"sources: {sources}\n" "---\n" "\nBody.\n" ) # --- the rendered form ------------------------------------------------------ def test_one_source_renders_the_byte_identical_single_form() -> None: """The additivity arm. Every golden in the suite reads this line.""" assert _render_sources([CATALOGUE]) == "[{ id: golden-catalogue, resource: fixture }]" def test_two_sources_render_as_one_flow_sequence_on_one_line() -> None: assert _render_sources([CATALOGUE, DATABASE]) == ( "[{ id: golden-catalogue, resource: fixture }, " "{ id: golden-db, resource: OKF_GOLDEN_SQL_DB }]" ) assert "\n" not in _render_sources([CATALOGUE, DATABASE]) def test_sources_keep_the_order_they_were_given() -> None: """No sort. The order a manifest names its sources in is the manifest's statement, and nothing here can recover it once reordered.""" forward = _render_sources([CATALOGUE, DATABASE, API]) reverse = _render_sources([API, DATABASE, CATALOGUE]) assert forward.index("golden-catalogue") < forward.index("golden-api") assert reverse.index("golden-api") < reverse.index("golden-catalogue") def test_an_empty_source_list_is_refused() -> None: """`sources: []` is a provenance record naming no source: it reads as a measured absence when it is the absence of a measurement.""" with pytest.raises(MaterializationError) as excinfo: _render_sources([]) assert excinfo.value.code == "sources_empty" # --- the round trip, which is the acceptance test --------------------------- def test_two_sources_round_trip_through_our_own_parser(tmp_path: Path) -> None: rendered = _render_sources([CATALOGUE, DATABASE]) path = tmp_path / "concept.md" path.write_text(_document(rendered), encoding="utf-8") frontmatter = parse_frontmatter(path) assert frontmatter["sources"] == rendered assert sorted(frontmatter) == ["generated", "sources", "title", "type"] def test_the_block_form_round_trips_to_nothing(tmp_path: Path) -> None: """The NEGATIVE CONTROL, and the measured reason the block form is not emitted. Two entries go in; an empty string comes back, and no error is raised anywhere. This test must stay green: it is the evidence, not a regression guard.""" path = tmp_path / "concept.md" path.write_text( "---\n" "type: dataset\n" "sources:\n" " - id: golden-catalogue\n" " resource: fixture\n" " - id: golden-db\n" " resource: OKF_GOLDEN_SQL_DB\n" "---\n" "\nBody.\n", encoding="utf-8", ) frontmatter = parse_frontmatter(path) assert frontmatter["sources"] == "" assert "golden-db" not in "".join(frontmatter.values()) # --- the refusal applies to every entry, not only the first ----------------- @pytest.mark.parametrize("position", [0, 1, 2], ids=["first", "middle", "last"]) def test_an_unquotable_locator_is_refused_in_any_position(position: int) -> None: """A gate that only reads the head of a list is a gate the second entry walks past.""" sources: list[Source] = [CATALOGUE, DATABASE, API] sources[position] = FileSource(id="broken", root="data, backup") with pytest.raises(MaterializationError) as excinfo: _render_sources(sources) assert excinfo.value.code == "source_reference_unquotable" @pytest.mark.parametrize("position", [0, 1], ids=["first", "last"]) def test_an_unquotable_id_is_refused_in_any_position(position: int) -> None: sources: list[Source] = [CATALOGUE, DATABASE] sources[position] = FileSource(id="a{1}", root="fixture") with pytest.raises(MaterializationError) as excinfo: _render_sources(sources) assert excinfo.value.code == "source_reference_unquotable"