PM decision B6 asked for a list-taking _render_sources so a concept can record more than one source, and prescribed the block list as the emitted form. The list is delivered; the block form is not. Three measurements, not an argument. Our own parse_frontmatter skips indented lines, so a block list round-trips to an empty value with every entry silently gone -- and _is_ingest_owned reads through that same parser. The consumer B6 was written for accepts the multi-entry flow sequence and classifies a block sequence as unreadable provenance, so block would hand it exactly the state it cannot read. And B6's own acceptance test asks for a round trip through this parser, which no block form can pass. A single source renders byte-identically, so all six goldens are unmoved. The unquotable-value gate now runs on every entry, not just the first. New code sources_empty refuses an empty list. 1023 -> 1034 tests, including the negative control that pins the block form's silent data loss.
152 lines
5.8 KiB
Python
152 lines
5.8 KiB
Python
"""SPEC 5.1 `sources` with more than one entry, and the form it is emitted in.
|
|
|
|
PM decision B6 asked for a list-taking `_render_sources` so a concept can
|
|
record more than one source. It also prescribed the BLOCK list as the emitted
|
|
form. The list is delivered; the block form is not, and this file carries the
|
|
measurement rather than the argument.
|
|
|
|
Three facts, each pinned by a test below:
|
|
|
|
- **Our own parser loses a block list entirely.** `parse_frontmatter` is
|
|
line-oriented and skips indented lines, so `sources:` followed by ` - id: a`
|
|
round-trips to an EMPTY value with every entry gone -- silently. A provenance
|
|
record we cannot read back is worse than one we never wrote.
|
|
- **The consumer's decoder reads the flow sequence and refuses the block one.**
|
|
`portfolio-optimiser`'s `decode_flow_value` accepts `[{ k: v }, { k: v }]` --
|
|
plural -- and classifies a block sequence as `UnreadableProvenance`. Emitting
|
|
block would hand the consumer that asked for multi-source exactly the state it
|
|
reports as unreadable.
|
|
- **The order's own acceptance test settles it.** It asks for a round trip
|
|
through our `parse_frontmatter` equivalent. No block form can pass that.
|
|
|
|
So the goal is delivered in the form that reaches a reader: one flow sequence of
|
|
N flow mappings, on one line. A single source stays byte-identical, which is
|
|
what keeps all six goldens unmoved.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from llm_ingestion_okf.errors import MaterializationError
|
|
from llm_ingestion_okf.manifest import FileSource, HttpSource, Source, SqlSource
|
|
from llm_ingestion_okf.materialize import _render_sources, parse_frontmatter
|
|
|
|
CATALOGUE = FileSource(id="golden-catalogue", root="fixture")
|
|
DATABASE = SqlSource(id="golden-db", connection_ref="OKF_GOLDEN_SQL_DB")
|
|
API = HttpSource(id="golden-api", base_url="https://golden.example.test")
|
|
|
|
|
|
def _document(sources: str) -> str:
|
|
return (
|
|
"---\n"
|
|
"type: dataset\n"
|
|
"title: Orders\n"
|
|
"generated: { by: process:okf-ingest, at: 2026-01-01T00:00:00Z }\n"
|
|
f"sources: {sources}\n"
|
|
"---\n"
|
|
"\nBody.\n"
|
|
)
|
|
|
|
|
|
# --- the rendered form ------------------------------------------------------
|
|
|
|
|
|
def test_one_source_renders_the_byte_identical_single_form() -> None:
|
|
"""The additivity arm. Every golden in the suite reads this line."""
|
|
assert _render_sources([CATALOGUE]) == "[{ id: golden-catalogue, resource: fixture }]"
|
|
|
|
|
|
def test_two_sources_render_as_one_flow_sequence_on_one_line() -> None:
|
|
assert _render_sources([CATALOGUE, DATABASE]) == (
|
|
"[{ id: golden-catalogue, resource: fixture }, "
|
|
"{ id: golden-db, resource: OKF_GOLDEN_SQL_DB }]"
|
|
)
|
|
assert "\n" not in _render_sources([CATALOGUE, DATABASE])
|
|
|
|
|
|
def test_sources_keep_the_order_they_were_given() -> None:
|
|
"""No sort. The order a manifest names its sources in is the manifest's
|
|
statement, and nothing here can recover it once reordered."""
|
|
forward = _render_sources([CATALOGUE, DATABASE, API])
|
|
reverse = _render_sources([API, DATABASE, CATALOGUE])
|
|
|
|
assert forward.index("golden-catalogue") < forward.index("golden-api")
|
|
assert reverse.index("golden-api") < reverse.index("golden-catalogue")
|
|
|
|
|
|
def test_an_empty_source_list_is_refused() -> None:
|
|
"""`sources: []` is a provenance record naming no source: it reads as a
|
|
measured absence when it is the absence of a measurement."""
|
|
with pytest.raises(MaterializationError) as excinfo:
|
|
_render_sources([])
|
|
|
|
assert excinfo.value.code == "sources_empty"
|
|
|
|
|
|
# --- the round trip, which is the acceptance test ---------------------------
|
|
|
|
|
|
def test_two_sources_round_trip_through_our_own_parser(tmp_path: Path) -> None:
|
|
rendered = _render_sources([CATALOGUE, DATABASE])
|
|
path = tmp_path / "concept.md"
|
|
path.write_text(_document(rendered), encoding="utf-8")
|
|
|
|
frontmatter = parse_frontmatter(path)
|
|
|
|
assert frontmatter["sources"] == rendered
|
|
assert sorted(frontmatter) == ["generated", "sources", "title", "type"]
|
|
|
|
|
|
def test_the_block_form_round_trips_to_nothing(tmp_path: Path) -> None:
|
|
"""The NEGATIVE CONTROL, and the measured reason the block form is not
|
|
emitted. Two entries go in; an empty string comes back, and no error is
|
|
raised anywhere. This test must stay green: it is the evidence, not a
|
|
regression guard."""
|
|
path = tmp_path / "concept.md"
|
|
path.write_text(
|
|
"---\n"
|
|
"type: dataset\n"
|
|
"sources:\n"
|
|
" - id: golden-catalogue\n"
|
|
" resource: fixture\n"
|
|
" - id: golden-db\n"
|
|
" resource: OKF_GOLDEN_SQL_DB\n"
|
|
"---\n"
|
|
"\nBody.\n",
|
|
encoding="utf-8",
|
|
)
|
|
|
|
frontmatter = parse_frontmatter(path)
|
|
|
|
assert frontmatter["sources"] == ""
|
|
assert "golden-db" not in "".join(frontmatter.values())
|
|
|
|
|
|
# --- the refusal applies to every entry, not only the first -----------------
|
|
|
|
|
|
@pytest.mark.parametrize("position", [0, 1, 2], ids=["first", "middle", "last"])
|
|
def test_an_unquotable_locator_is_refused_in_any_position(position: int) -> None:
|
|
"""A gate that only reads the head of a list is a gate the second entry
|
|
walks past."""
|
|
sources: list[Source] = [CATALOGUE, DATABASE, API]
|
|
sources[position] = FileSource(id="broken", root="data, backup")
|
|
|
|
with pytest.raises(MaterializationError) as excinfo:
|
|
_render_sources(sources)
|
|
|
|
assert excinfo.value.code == "source_reference_unquotable"
|
|
|
|
|
|
@pytest.mark.parametrize("position", [0, 1], ids=["first", "last"])
|
|
def test_an_unquotable_id_is_refused_in_any_position(position: int) -> None:
|
|
sources: list[Source] = [CATALOGUE, DATABASE]
|
|
sources[position] = FileSource(id="a{1}", root="fixture")
|
|
|
|
with pytest.raises(MaterializationError) as excinfo:
|
|
_render_sources(sources)
|
|
|
|
assert excinfo.value.code == "source_reference_unquotable"
|