feat(materialize): sources takes a list and renders N flow mappings
PM decision B6 asked for a list-taking _render_sources so a concept can record more than one source, and prescribed the block list as the emitted form. The list is delivered; the block form is not. Three measurements, not an argument. Our own parse_frontmatter skips indented lines, so a block list round-trips to an empty value with every entry silently gone -- and _is_ingest_owned reads through that same parser. The consumer B6 was written for accepts the multi-entry flow sequence and classifies a block sequence as unreadable provenance, so block would hand it exactly the state it cannot read. And B6's own acceptance test asks for a round trip through this parser, which no block form can pass. A single source renders byte-identically, so all six goldens are unmoved. The unquotable-value gate now runs on every entry, not just the first. New code sources_empty refuses an empty list. 1023 -> 1034 tests, including the negative control that pins the block form's silent data loss.
This commit is contained in:
parent
bc0b4130f1
commit
16eeeb007e
4 changed files with 240 additions and 32 deletions
152
tests/test_multi_source_provenance.py
Normal file
152
tests/test_multi_source_provenance.py
Normal file
|
|
@ -0,0 +1,152 @@
|
|||
"""SPEC 5.1 `sources` with more than one entry, and the form it is emitted in.
|
||||
|
||||
PM decision B6 asked for a list-taking `_render_sources` so a concept can
|
||||
record more than one source. It also prescribed the BLOCK list as the emitted
|
||||
form. The list is delivered; the block form is not, and this file carries the
|
||||
measurement rather than the argument.
|
||||
|
||||
Three facts, each pinned by a test below:
|
||||
|
||||
- **Our own parser loses a block list entirely.** `parse_frontmatter` is
|
||||
line-oriented and skips indented lines, so `sources:` followed by ` - id: a`
|
||||
round-trips to an EMPTY value with every entry gone -- silently. A provenance
|
||||
record we cannot read back is worse than one we never wrote.
|
||||
- **The consumer's decoder reads the flow sequence and refuses the block one.**
|
||||
`portfolio-optimiser`'s `decode_flow_value` accepts `[{ k: v }, { k: v }]` --
|
||||
plural -- and classifies a block sequence as `UnreadableProvenance`. Emitting
|
||||
block would hand the consumer that asked for multi-source exactly the state it
|
||||
reports as unreadable.
|
||||
- **The order's own acceptance test settles it.** It asks for a round trip
|
||||
through our `parse_frontmatter` equivalent. No block form can pass that.
|
||||
|
||||
So the goal is delivered in the form that reaches a reader: one flow sequence of
|
||||
N flow mappings, on one line. A single source stays byte-identical, which is
|
||||
what keeps all six goldens unmoved.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from llm_ingestion_okf.errors import MaterializationError
|
||||
from llm_ingestion_okf.manifest import FileSource, HttpSource, Source, SqlSource
|
||||
from llm_ingestion_okf.materialize import _render_sources, parse_frontmatter
|
||||
|
||||
CATALOGUE = FileSource(id="golden-catalogue", root="fixture")
|
||||
DATABASE = SqlSource(id="golden-db", connection_ref="OKF_GOLDEN_SQL_DB")
|
||||
API = HttpSource(id="golden-api", base_url="https://golden.example.test")
|
||||
|
||||
|
||||
def _document(sources: str) -> str:
|
||||
return (
|
||||
"---\n"
|
||||
"type: dataset\n"
|
||||
"title: Orders\n"
|
||||
"generated: { by: process:okf-ingest, at: 2026-01-01T00:00:00Z }\n"
|
||||
f"sources: {sources}\n"
|
||||
"---\n"
|
||||
"\nBody.\n"
|
||||
)
|
||||
|
||||
|
||||
# --- the rendered form ------------------------------------------------------
|
||||
|
||||
|
||||
def test_one_source_renders_the_byte_identical_single_form() -> None:
|
||||
"""The additivity arm. Every golden in the suite reads this line."""
|
||||
assert _render_sources([CATALOGUE]) == "[{ id: golden-catalogue, resource: fixture }]"
|
||||
|
||||
|
||||
def test_two_sources_render_as_one_flow_sequence_on_one_line() -> None:
|
||||
assert _render_sources([CATALOGUE, DATABASE]) == (
|
||||
"[{ id: golden-catalogue, resource: fixture }, "
|
||||
"{ id: golden-db, resource: OKF_GOLDEN_SQL_DB }]"
|
||||
)
|
||||
assert "\n" not in _render_sources([CATALOGUE, DATABASE])
|
||||
|
||||
|
||||
def test_sources_keep_the_order_they_were_given() -> None:
|
||||
"""No sort. The order a manifest names its sources in is the manifest's
|
||||
statement, and nothing here can recover it once reordered."""
|
||||
forward = _render_sources([CATALOGUE, DATABASE, API])
|
||||
reverse = _render_sources([API, DATABASE, CATALOGUE])
|
||||
|
||||
assert forward.index("golden-catalogue") < forward.index("golden-api")
|
||||
assert reverse.index("golden-api") < reverse.index("golden-catalogue")
|
||||
|
||||
|
||||
def test_an_empty_source_list_is_refused() -> None:
|
||||
"""`sources: []` is a provenance record naming no source: it reads as a
|
||||
measured absence when it is the absence of a measurement."""
|
||||
with pytest.raises(MaterializationError) as excinfo:
|
||||
_render_sources([])
|
||||
|
||||
assert excinfo.value.code == "sources_empty"
|
||||
|
||||
|
||||
# --- the round trip, which is the acceptance test ---------------------------
|
||||
|
||||
|
||||
def test_two_sources_round_trip_through_our_own_parser(tmp_path: Path) -> None:
|
||||
rendered = _render_sources([CATALOGUE, DATABASE])
|
||||
path = tmp_path / "concept.md"
|
||||
path.write_text(_document(rendered), encoding="utf-8")
|
||||
|
||||
frontmatter = parse_frontmatter(path)
|
||||
|
||||
assert frontmatter["sources"] == rendered
|
||||
assert sorted(frontmatter) == ["generated", "sources", "title", "type"]
|
||||
|
||||
|
||||
def test_the_block_form_round_trips_to_nothing(tmp_path: Path) -> None:
|
||||
"""The NEGATIVE CONTROL, and the measured reason the block form is not
|
||||
emitted. Two entries go in; an empty string comes back, and no error is
|
||||
raised anywhere. This test must stay green: it is the evidence, not a
|
||||
regression guard."""
|
||||
path = tmp_path / "concept.md"
|
||||
path.write_text(
|
||||
"---\n"
|
||||
"type: dataset\n"
|
||||
"sources:\n"
|
||||
" - id: golden-catalogue\n"
|
||||
" resource: fixture\n"
|
||||
" - id: golden-db\n"
|
||||
" resource: OKF_GOLDEN_SQL_DB\n"
|
||||
"---\n"
|
||||
"\nBody.\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
frontmatter = parse_frontmatter(path)
|
||||
|
||||
assert frontmatter["sources"] == ""
|
||||
assert "golden-db" not in "".join(frontmatter.values())
|
||||
|
||||
|
||||
# --- the refusal applies to every entry, not only the first -----------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize("position", [0, 1, 2], ids=["first", "middle", "last"])
|
||||
def test_an_unquotable_locator_is_refused_in_any_position(position: int) -> None:
|
||||
"""A gate that only reads the head of a list is a gate the second entry
|
||||
walks past."""
|
||||
sources: list[Source] = [CATALOGUE, DATABASE, API]
|
||||
sources[position] = FileSource(id="broken", root="data, backup")
|
||||
|
||||
with pytest.raises(MaterializationError) as excinfo:
|
||||
_render_sources(sources)
|
||||
|
||||
assert excinfo.value.code == "source_reference_unquotable"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("position", [0, 1], ids=["first", "last"])
|
||||
def test_an_unquotable_id_is_refused_in_any_position(position: int) -> None:
|
||||
sources: list[Source] = [CATALOGUE, DATABASE]
|
||||
sources[position] = FileSource(id="a{1}", root="fixture")
|
||||
|
||||
with pytest.raises(MaterializationError) as excinfo:
|
||||
_render_sources(sources)
|
||||
|
||||
assert excinfo.value.code == "source_reference_unquotable"
|
||||
Loading…
Add table
Add a link
Reference in a new issue