"""Cut an OKF bundle to one contract-conformant payload for one question. This is the **pre-pass** `docs/consumption-contract.md` SS 1 names: the deterministic program that reads the bundle, ranks its concepts, cuts them to a bounded set, and emits one payload. It decides nothing about the question being asked -- the skill does the judgement, this does the reading, the ranking and the cut (SS 2.1). **What it deliberately does not do, and the failure each refusal prevents:** - **It calls no model.** A model in the run path makes the same question at the same ref return different bytes, which is the whole property a declared cut buys over an emergent one. - **It opens no socket.** The repository requires an explicit per-run opt-in for network access; this command takes no such flag, so it can never reach one. - **It enumerates no directory.** SS 9.2 forbids that unless the named profile says the index is derived, and measured 2026-09-07, `entries_match_directory` is `True` for `STRICT_V1` alone -- for none of the profiles a segmented v0.2 bundle could have been built under. So the walk follows the INDEX TREE. That costs nothing: measured on the 629-concept K2 bundle, the index walk reaches exactly the set `rglob` finds. - **It imports nothing outside the standard library and this repository.** The package pins exactly one runtime dependency and a packaging test enforces it. It lives outside `src/`, so it never enters a wheel and no consumer's install surface changes because it exists -- the reason `tools/okf_contract_check.py` states for its own location. The entry point is `build_payload(...)`, with the CLI a thin `main()`, so lifting it into `src/` the day a consumer asks for a wheel-installed command is a move rather than a rewrite. """ from __future__ import annotations import hashlib import sys from collections.abc import Mapping from dataclasses import dataclass from pathlib import Path from typing import Literal sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) from llm_ingestion_okf.inbox import ( # noqa: E402 ADJUDICATION_ADJUDICATED, ADJUDICATION_PROPOSED, ADJUDICATION_STATES, ) from llm_ingestion_okf.materialize import parse_frontmatter # noqa: E402 from llm_ingestion_okf.profiles import SEGMENTED_OKF_V0_2, BundleProfile # noqa: E402 #: The profile whose index policy reads a faceted, per-directory index -- the #: shape both the in-repo golden bundle and the K2 corpus carry. Named as this #: instrument's default rather than hard-coded at each call site: a caller with #: a differently-shaped bundle passes its own. DEFAULT_PROFILE = SEGMENTED_OKF_V0_2 #: The algorithm the `ref` names, spelled in the value itself. SS 3.3 requires #: "a commit or equivalent content identity" and names no algorithm; a bundle #: that is not a git checkout has no commit, so the identity is computed. Naming #: the algorithm inline is what lets a consumer reproduce it without a document. REF_ALGORITHM = "sha256-tree" #: The one concept suffix this instrument reads, taken from the profile so a #: second literal cannot drift from it. CONCEPT_SUFFIX = DEFAULT_PROFILE.paths.concept_suffix def _walk_index_tree( bundle_root: Path, *, profile: BundleProfile ) -> tuple[tuple[str, ...], tuple[str, ...]]: """Every index file and every concept id the index tree reaches. One walk, two results, because the alternative is two walks that can disagree -- and a ref computed over a different set than the one that was read is an identity for something nobody consumed. """ index_name = profile.index.name suffix = profile.paths.concept_suffix indexes: set[str] = set() concepts: set[str] = set() pending: list[str] = [index_name] while pending: relative = pending.pop() if relative in indexes: continue indexes.add(relative) index_file = bundle_root / relative if not index_file.is_file(): continue parent = _parent_dir(relative) for line in index_file.read_text(encoding="utf-8").splitlines(): entry = profile.index.parse_entry(line) # `None` is curated prose, not a defect: the index is the one file # where this library writes beside somebody else's text. if entry is None: continue target = _join(parent, entry.target) if target is None: continue if target.rsplit("/", 1)[-1] == index_name: pending.append(target) elif target.endswith(suffix): concepts.add(target[: -len(suffix)]) return _byte_sorted(indexes), _byte_sorted(concepts) def _byte_sorted(values: set[str]) -> tuple[str, ...]: """Sorted by the UTF-8 encoding, never by the locale. A locale-dependent order makes byte-identical output a claim about the machine rather than about the bundle. """ return tuple(sorted(values, key=lambda value: value.encode("utf-8"))) def enumerate_concepts( bundle_root: Path, *, profile: BundleProfile = DEFAULT_PROFILE ) -> tuple[str, ...]: """Every concept id in the bundle, reached through the index tree. Ids are bundle-relative POSIX paths with the profile's concept suffix removed -- slash-preserving, so two same-named concepts in different documents stay two concepts. **Not `rglob`.** SS 9.2 forbids the consumer path from enumerating a directory unless the named profile says the index is derived. The index is safe to rely on here because Door B recomputes it as a projection over the whole bundle each round, which is what makes a rebuild from scratch equal an incremental update byte for byte. """ _, concepts = _walk_index_tree(bundle_root, profile=profile) return concepts def _parent_dir(relative: str) -> str: """The directory part of a bundle-relative POSIX path, or `""` at the root. Spelled here rather than as `PurePosixPath(...).parent` so the root case is `""` and not `"."` -- joining `"."` would put a `./` segment into every id at depth one, and two ids differing only by that segment are two ids for one concept. """ head, sep, _ = relative.rpartition("/") return head if sep else "" def _join(parent: str, target: str) -> str | None: """A relative index target resolved against the index's own directory. Returns `None` for a target that climbs above the bundle root or is absolute. A target escaping the root is refused rather than clamped: a clamped path names a real file the bundle never pointed at. """ if target.startswith("/"): return None parts: list[str] = parent.split("/") if parent else [] for segment in target.split("/"): if segment in ("", "."): continue if segment == "..": if not parts: return None parts.pop() continue parts.append(segment) return "/".join(parts) if parts else None def bundle_ref(bundle_root: Path, *, profile: BundleProfile = DEFAULT_PROFILE) -> str: """A content identity for the bundle: `sha256-tree:`. The digest is taken over the LF-joined, byte-sorted lines `\t` across every file the INDEX TREE reaches -- the indexes themselves and the concepts they name. **Reachable, not `rglob`, and the cost is stated rather than hidden.** A file in the directory that no index names is outside this identity. It is also outside what the pre-pass may read (SS 9.2), so an identity covering it would be an assertion about bytes this command is forbidden to look at. The property that matters holds: every byte that can reach a payload is inside the ref, so the ref moves whenever a delivered excerpt could. **Path and bytes, never mtime or size.** A digest over metadata would move on a copy and hold still on an edit that preserved length, which is the opposite of what an identity is for. """ indexes, concepts = _walk_index_tree(bundle_root, profile=profile) suffix = profile.paths.concept_suffix lines: list[bytes] = [] for relative in (*indexes, *(f"{concept}{suffix}" for concept in concepts)): path = bundle_root / relative if not path.is_file(): continue digest = hashlib.sha256(path.read_bytes()).hexdigest() lines.append(f"{relative}\t{digest}".encode()) joined = b"\n".join(sorted(lines)) return f"{REF_ALGORITHM}:{hashlib.sha256(joined).hexdigest()}" class ConsumeError(Exception): """A refusal this instrument can name. Carries a `code` for the same reason `okf_contract_check.Finding` does: one "invalid" verdict over a dozen defects is a diagnostic no caller can act on. """ def __init__(self, message: str, *, code: str) -> None: super().__init__(message) self.code = code #: The three states a CONSUMER must distinguish (SS 6.1), against the two the #: WIRE carries (`inbox.ADJUDICATION_STATES`). The difference is the whole #: point: `unknown` is not a value a producer writes, it is what the absence of #: the key means, and it is written explicitly here so a reader never has to #: infer it from a missing member. CONSUMER_ADJUDICATION_STATES = (*ADJUDICATION_STATES, "unknown") AdjudicationState = Literal["proposed", "adjudicated", "unknown"] @dataclass(frozen=True) class Concept: """One concept read off disk, with every state written rather than implied.""" path: Path concept_id: str bundle_id: str #: `True` when `bundle_id` came from the root index rather than the concept. #: Recorded rather than silently defaulted: SS 3.1 makes identity the #: `(bundle_id, concept_id)` tuple, so where the first half came from is #: part of what the payload is asserting. bundle_id_inherited: bool sha256: str okf_type: str title: str source_file: str adjudication: AdjudicationState #: `False` when the key was absent. `adjudication == "unknown"` already says #: so, but a separate flag keeps the two facts from being one inference. adjudication_present: bool frontmatter: Mapping[str, str] body: str def read_concept(path: Path, *, bundle_root: Path, root_bundle_id: str) -> Concept: """One concept file as a record. Reads; derives nothing about the question. `sha256` is the digest of the CONCEPT FILE (SS 3.2), never the `source_sha256` frontmatter key -- that one digests the source document the concept was extracted from, and conflating them would make the payload's content identity point at a PDF nobody in the chain reads. Both exist on every K2 concept, which is what makes the confusion available. """ frontmatter = parse_frontmatter(path) relative = path.relative_to(bundle_root).as_posix() #: The slash-preserving id: the bundle-relative path minus the suffix. This #: instrument's own choice, consistent with the RULE at `importer.py:244` #: and `inbox.py:1101-1102` -- but deliberately NOT `importer.import_slug`, #: which one line further down flattens the id to a single hyphenated #: segment. A flattened id fails a document-prefix match in a way that looks #: like a ranking miss rather than an id-format bug. concept_id = relative[: -len(CONCEPT_SUFFIX)] if relative.endswith(CONCEPT_SUFFIX) else relative raw = frontmatter.get("adjudication") if raw is None: adjudication: AdjudicationState = "unknown" elif raw == ADJUDICATION_PROPOSED: adjudication = "proposed" elif raw == ADJUDICATION_ADJUDICATED: adjudication = "adjudicated" else: raise ConsumeError( f"{relative} carries adjudication={raw!r}, outside the wire set " f"{ADJUDICATION_STATES}; refusing to map it to 'unknown', which " "would report 'we cannot tell whether it was judged' where the " "truth is that the bundle said something this consumer does not " "understand", code="adjudication_unknown_value", ) declared = frontmatter.get("bundle_id") return Concept( path=path, concept_id=concept_id, bundle_id=declared if declared else root_bundle_id, bundle_id_inherited=not declared, sha256=hashlib.sha256(path.read_bytes()).hexdigest(), okf_type=frontmatter.get("type", ""), title=frontmatter.get("title", ""), source_file=frontmatter.get("source_file", ""), adjudication=adjudication, adjudication_present=raw is not None, frontmatter=frontmatter, body=_body(path), ) def _body(path: Path) -> str: """The text after the frontmatter block, or the whole file when there is none.""" text = path.read_text(encoding="utf-8") lines = text.splitlines() if not lines or lines[0].strip() != "---": return text for offset, line in enumerate(lines[1:], start=2): if line.strip() == "---": return "\n".join(lines[offset:]) return text #: SS 6.2, from SPEC SS 5.3, lowest to highest. Imported from the checker's own #: constant would be circular (the checker is a separate tool); spelled here and #: held to the checker's set by the anti-drift test. TrustTier = Literal["unverified", "machine-confirmed", "human-reviewed"] #: The prefix that makes an actor a person. A PREFIX, never a substring: an #: actor id `bot/human:2` contains the literal and is a machine, and promoting #: it would be fabricated provenance produced by a matching bug. HUMAN_ACTOR_PREFIX = "human:" def trust_tier(verified_raw: str | None) -> TrustTier | None: """The tier `verified` implies, or `None` when the value cannot be read. Three inputs, three different facts, and collapsing any two is the defect: - `None` -- the key is ABSENT. SS 6.2: no `verified` key means `unverified`, and SS 6.3 forbids rejecting a concept for it. - `""` -- the key is PRESENT and this library cannot read it. Measured 2026-09-07: the line-oriented `parse_frontmatter` returns `''` for a block-form value and the full string for a flow one, so the two are distinguishable. Returning `None` here is not a tier; the caller withholds the concept under a named rule. Emitting `unverified` instead would assert a fact nobody measured, which is exactly what SS 6.4 forbids. - a flow value -- decoded, and the tier follows the actors. """ if verified_raw is None: return "unverified" if not verified_raw.strip(): return None entries = _parse_flow_mappings(verified_raw) if entries is None: return None actors: list[str] = [] for entry in entries: actor = entry.get("by") if not actor: raise ConsumeError( f"a `verified` entry names no `by` actor ({verified_raw!r}); " "refusing to tier it, because the tier IS a claim about who " "checked and there is nobody to name", code="verified_actorless", ) actors.append(actor) if not actors: return None if any(actor.startswith(HUMAN_ACTOR_PREFIX) for actor in actors): return "human-reviewed" return "machine-confirmed" def _parse_flow_mappings(value: str) -> list[dict[str, str]] | None: """A YAML flow sequence of flow mappings, or `None` when it is not one. Modelled on `structure._parse_flow_list` -- flow form only, because this library's standing rule is that a value it can write is a value it can read back. Not reused: that one splits on every comma, and `{ by: x, at: y }` carries a comma INSIDE a mapping, so it would return four fragments where there are two pairs. """ stripped = value.strip() if not (stripped.startswith("[") and stripped.endswith("]")): return None body = stripped[1:-1].strip() if not body: return [] mappings: list[dict[str, str]] = [] for chunk in _split_top_level(body, "{", "}"): item = chunk.strip() if not (item.startswith("{") and item.endswith("}")): return None pairs: dict[str, str] = {} for field in item[1:-1].split(","): key, separator, raw = field.partition(":") if separator: pairs[key.strip()] = raw.strip() mappings.append(pairs) return mappings def _split_top_level(body: str, opener: str, closer: str) -> list[str]: """Split on commas that are not inside a `{...}`.""" parts: list[str] = [] depth = 0 current: list[str] = [] for character in body: if character == opener: depth += 1 elif character == closer: depth -= 1 if character == "," and depth == 0: parts.append("".join(current)) current = [] continue current.append(character) parts.append("".join(current)) return [part for part in parts if part.strip()]