feat(okf): decode_flow_value reads the accepted flow subset and refuses the rest by name

Co-Authored-By: Claude <claude-opus-5>
This commit is contained in:
Kjell Tore Guttormsen 2026-09-02 20:14:00 +02:00
commit 940796d9ed
2 changed files with 338 additions and 0 deletions

View file

@ -129,6 +129,208 @@ def _read_body(path: Path) -> str:
return _body_from_text(path.read_text(encoding="utf-8"))
class FlowDecodeError(ValueError):
"""A frontmatter value is outside the accepted single-line flow subset.
A ``ValueError`` subclass on purpose (the ``IngestStampError`` precedent): the CLI's refusal
tuple and hosting's 400 arm both catch ``ValueError``, so a malformed knowledge base is a
refusal the caller can read, never a traceback on the crash channel."""
#: The pair separator INSIDE a flow mapping. Colon-SPACE, never a bare colon — ``by: human:jsmith``
#: and ``at: 2024-01-15T10:00:00Z`` both carry colons that are part of the value, and a decoder
#: that split on ``:`` would quietly truncate every actor and every timestamp it read.
_FLOW_PAIR_SEPARATOR = ": "
def _scan_flow(text: str, separator: str) -> tuple[list[str], str, int]:
"""Split ``text`` on ``separator`` at nesting depth 0 and OUTSIDE quotes.
Returns ``(parts, unclosed_quote, depth)`` the two trailing values are how the caller tells a
complete value from a truncated one, rather than discovering it later as a wrong answer.
"Split on a separator" is where this class of decoder fails silently, so the scan is
character-by-character and quote-aware instead of ``str.split``."""
parts: list[str] = []
buf: list[str] = []
quote = ""
depth = 0
i = 0
while i < len(text):
ch = text[i]
if quote:
buf.append(ch)
if ch == quote:
quote = ""
i += 1
continue
if ch in "\"'":
quote = ch
buf.append(ch)
i += 1
continue
if ch in "[{":
depth += 1
buf.append(ch)
i += 1
continue
if ch in "]}":
depth -= 1
buf.append(ch)
i += 1
continue
if depth == 0 and text.startswith(separator, i):
parts.append("".join(buf))
buf = []
i += len(separator)
continue
buf.append(ch)
i += 1
parts.append("".join(buf))
return parts, quote, depth
def _has_nested_collection(text: str) -> bool:
"""True when ``text`` opens a ``[`` or ``{`` outside quotes. Nesting is outside the accepted
subset, and depth alone cannot detect it a balanced ``[x, y]`` returns to depth 0."""
quote = ""
for ch in text:
if quote:
if ch == quote:
quote = ""
continue
if ch in "\"'":
quote = ch
continue
if ch in "[{":
return True
return False
def _find_pair_separator(pair: str) -> int:
"""Index of the FIRST ``": "`` outside quotes, or ``-1``."""
quote = ""
i = 0
while i < len(pair):
ch = pair[i]
if quote:
if ch == quote:
quote = ""
i += 1
continue
if ch in "\"'":
quote = ch
i += 1
continue
if pair.startswith(_FLOW_PAIR_SEPARATOR, i):
return i
i += 1
return -1
def _decode_flow_mapping(item: str, raw: str, key: str | None) -> dict[str, str]:
"""One ``{ k: v, ... }`` flow mapping into a dict, KEY-AGNOSTICALLY."""
inner = item[1:-1]
if _has_nested_collection(inner):
raise FlowDecodeError(
f"a nested flow collection inside {raw!r} is outside the accepted subset — the "
"decoder reads one level of `{ key: value }` pairs and refuses to guess at more"
)
pairs, quote, depth = _scan_flow(inner, ",")
if quote:
raise FlowDecodeError(f"an unterminated quoted scalar in {raw!r}")
if depth != 0:
raise FlowDecodeError(f"an unterminated flow mapping in {raw!r}")
entry: dict[str, str] = {}
for pair in pairs:
at = _find_pair_separator(pair)
if at < 0:
raise FlowDecodeError(
f"{pair.strip()!r} in {raw!r} is not a `key: value` pair (the separator is "
"colon-SPACE) — refusing rather than guessing what was meant"
)
name = unquote_scalar(pair[:at])
value = unquote_scalar(pair[at + len(_FLOW_PAIR_SEPARATOR) :])
if name in entry:
raise FlowDecodeError(
f"duplicate key {name!r} in {raw!r} — last-write-wins is precisely the silent "
"overwrite this decoder exists to remove, so it is refused here too"
)
entry[name] = value
if key == "verified" and not entry.get("by"):
raise FlowDecodeError(
f"a `verified` entry in {raw!r} names no actor — SPEC §5.2 makes `by` required within "
"a verification event, and tiering an entry that names nobody would mint provenance"
)
return entry
def decode_flow_value(raw: str, *, key: str | None = None) -> tuple[dict[str, str], ...]:
"""Decode the accepted single-line flow subset into a tuple of entries.
Two shapes are accepted and nothing else: a flow sequence of flow mappings
``[{ k: v }, { k: v }]``, and a bare flow mapping ``{ k: v }`` which normalises to a
ONE-ELEMENT tuple (SPEC §5.2: "Consumers MUST treat a bare mapping as a one-element list").
Entries decode **key-agnostically** whatever keys the entry carries, never a hard-coded
``{id, resource}``. The agreed segmented shape adds ``segment_id`` and ``source_offset``, so a
two-key decoder would refuse the very bundles this seam is built for. This is the ONE named
seam: a future structured reader becomes a parameter here, not a refactor everywhere.
**No YAML-1.1 coercion.** Values come back as the strings they were written as: ``yes`` / ``no``
/ ``on`` stay strings and ``1`` stays ``"1"``. This is a DELIBERATE divergence from PyYAML's
resolver, which would return ``True`` and ``1`` written down here rather than inherited
silently, because a value that changes type between the file and the consumer is exactly the
class of surprise a provenance reader must not import.
``key`` is optional and additive: it carries the ONE key-specific rule SPEC §5.2 imposes, that
a ``verified`` entry must name an actor. Everything else stays key-agnostic.
Raises ``FlowDecodeError`` by name, with its own message for a bare-scalar sequence, an
empty sequence, an unterminated flow, a value continued onto the next line, a nested collection,
a duplicate key within one entry, and a ``verified`` entry with no ``by``.
Gated by ``tests/test_provenance_decoder_loadbearing.py``."""
if "\n" in raw or "\r" in raw:
raise FlowDecodeError(
f"the flow value {raw!r} does not fit on one line — a continued value is outside the "
"accepted subset, and joining the lines would decode something nobody wrote"
)
text = raw.strip()
if text.startswith("["):
if not text.endswith("]"):
raise FlowDecodeError(f"an unterminated flow sequence: {raw!r}")
inner = text[1:-1].strip()
if not inner:
raise FlowDecodeError(
f"the flow sequence {raw!r} names no source — an empty list reads as a measured "
"absence when it is the absence of a measurement"
)
chunks, quote, depth = _scan_flow(inner, ",")
if quote:
raise FlowDecodeError(f"an unterminated quoted scalar in {raw!r}")
if depth != 0:
raise FlowDecodeError(f"an unterminated flow mapping in {raw!r}")
entries: list[dict[str, str]] = []
for chunk in chunks:
item = chunk.strip()
if not (item.startswith("{") and item.endswith("}")):
raise FlowDecodeError(
f"a bare scalar entry {item!r} in {raw!r} is not a conformant entry — SPEC "
"§5.1 makes `resource` REQUIRED within an entry, so a naked filename names a "
"resource without saying so"
)
entries.append(_decode_flow_mapping(item, raw, key))
return tuple(entries)
if text.startswith("{"):
if not text.endswith("}"):
raise FlowDecodeError(f"an unterminated flow mapping: {raw!r}")
return (_decode_flow_mapping(text, raw, key),)
raise FlowDecodeError(
f"{raw!r} is neither a flow sequence nor a flow mapping — the accepted subset is "
"`[{ k: v }, ...]` or `{ k: v }` on one line"
)
@dataclass(frozen=True)
class BundleFile:
"""One OKF file: its name, declared ``type`` (``""`` if absent), frontmatter, and body."""