184 lines
8 KiB
Python
184 lines
8 KiB
Python
"""Cut an OKF bundle to one contract-conformant payload for one question.
|
|
|
|
This is the **pre-pass** `docs/consumption-contract.md` SS 1 names: the
|
|
deterministic program that reads the bundle, ranks its concepts, cuts them to a
|
|
bounded set, and emits one payload. It decides nothing about the question being
|
|
asked -- the skill does the judgement, this does the reading, the ranking and
|
|
the cut (SS 2.1).
|
|
|
|
**What it deliberately does not do, and the failure each refusal prevents:**
|
|
|
|
- **It calls no model.** A model in the run path makes the same question at the
|
|
same ref return different bytes, which is the whole property a declared cut
|
|
buys over an emergent one.
|
|
- **It opens no socket.** The repository requires an explicit per-run opt-in for
|
|
network access; this command takes no such flag, so it can never reach one.
|
|
- **It enumerates no directory.** SS 9.2 forbids that unless the named profile
|
|
says the index is derived, and measured 2026-09-07, `entries_match_directory`
|
|
is `True` for `STRICT_V1` alone -- for none of the profiles a segmented v0.2
|
|
bundle could have been built under. So the walk follows the INDEX TREE. That
|
|
costs nothing: measured on the 629-concept K2 bundle, the index walk reaches
|
|
exactly the set `rglob` finds.
|
|
- **It imports nothing outside the standard library and this repository.** The
|
|
package pins exactly one runtime dependency and a packaging test enforces it.
|
|
|
|
It lives outside `src/`, so it never enters a wheel and no consumer's install
|
|
surface changes because it exists -- the reason `tools/okf_contract_check.py`
|
|
states for its own location. The entry point is `build_payload(...)`, with the
|
|
CLI a thin `main()`, so lifting it into `src/` the day a consumer asks for a
|
|
wheel-installed command is a move rather than a rewrite.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
|
|
|
|
from llm_ingestion_okf.profiles import SEGMENTED_OKF_V0_2, BundleProfile # noqa: E402
|
|
|
|
#: The profile whose index policy reads a faceted, per-directory index -- the
|
|
#: shape both the in-repo golden bundle and the K2 corpus carry. Named as this
|
|
#: instrument's default rather than hard-coded at each call site: a caller with
|
|
#: a differently-shaped bundle passes its own.
|
|
DEFAULT_PROFILE = SEGMENTED_OKF_V0_2
|
|
|
|
#: The algorithm the `ref` names, spelled in the value itself. SS 3.3 requires
|
|
#: "a commit or equivalent content identity" and names no algorithm; a bundle
|
|
#: that is not a git checkout has no commit, so the identity is computed. Naming
|
|
#: the algorithm inline is what lets a consumer reproduce it without a document.
|
|
REF_ALGORITHM = "sha256-tree"
|
|
|
|
|
|
def _walk_index_tree(
|
|
bundle_root: Path, *, profile: BundleProfile
|
|
) -> tuple[tuple[str, ...], tuple[str, ...]]:
|
|
"""Every index file and every concept id the index tree reaches.
|
|
|
|
One walk, two results, because the alternative is two walks that can
|
|
disagree -- and a ref computed over a different set than the one that was
|
|
read is an identity for something nobody consumed.
|
|
"""
|
|
index_name = profile.index.name
|
|
suffix = profile.paths.concept_suffix
|
|
indexes: set[str] = set()
|
|
concepts: set[str] = set()
|
|
pending: list[str] = [index_name]
|
|
while pending:
|
|
relative = pending.pop()
|
|
if relative in indexes:
|
|
continue
|
|
indexes.add(relative)
|
|
index_file = bundle_root / relative
|
|
if not index_file.is_file():
|
|
continue
|
|
parent = _parent_dir(relative)
|
|
for line in index_file.read_text(encoding="utf-8").splitlines():
|
|
entry = profile.index.parse_entry(line)
|
|
# `None` is curated prose, not a defect: the index is the one file
|
|
# where this library writes beside somebody else's text.
|
|
if entry is None:
|
|
continue
|
|
target = _join(parent, entry.target)
|
|
if target is None:
|
|
continue
|
|
if target.rsplit("/", 1)[-1] == index_name:
|
|
pending.append(target)
|
|
elif target.endswith(suffix):
|
|
concepts.add(target[: -len(suffix)])
|
|
return _byte_sorted(indexes), _byte_sorted(concepts)
|
|
|
|
|
|
def _byte_sorted(values: set[str]) -> tuple[str, ...]:
|
|
"""Sorted by the UTF-8 encoding, never by the locale.
|
|
|
|
A locale-dependent order makes byte-identical output a claim about the
|
|
machine rather than about the bundle.
|
|
"""
|
|
return tuple(sorted(values, key=lambda value: value.encode("utf-8")))
|
|
|
|
|
|
def enumerate_concepts(
|
|
bundle_root: Path, *, profile: BundleProfile = DEFAULT_PROFILE
|
|
) -> tuple[str, ...]:
|
|
"""Every concept id in the bundle, reached through the index tree.
|
|
|
|
Ids are bundle-relative POSIX paths with the profile's concept suffix
|
|
removed -- slash-preserving, so two same-named concepts in different
|
|
documents stay two concepts.
|
|
|
|
**Not `rglob`.** SS 9.2 forbids the consumer path from enumerating a
|
|
directory unless the named profile says the index is derived. The index is
|
|
safe to rely on here because Door B recomputes it as a projection over the
|
|
whole bundle each round, which is what makes a rebuild from scratch equal an
|
|
incremental update byte for byte.
|
|
"""
|
|
_, concepts = _walk_index_tree(bundle_root, profile=profile)
|
|
return concepts
|
|
|
|
|
|
def _parent_dir(relative: str) -> str:
|
|
"""The directory part of a bundle-relative POSIX path, or `""` at the root.
|
|
|
|
Spelled here rather than as `PurePosixPath(...).parent` so the root case is
|
|
`""` and not `"."` -- joining `"."` would put a `./` segment into every id
|
|
at depth one, and two ids differing only by that segment are two ids for one
|
|
concept.
|
|
"""
|
|
head, sep, _ = relative.rpartition("/")
|
|
return head if sep else ""
|
|
|
|
|
|
def _join(parent: str, target: str) -> str | None:
|
|
"""A relative index target resolved against the index's own directory.
|
|
|
|
Returns `None` for a target that climbs above the bundle root or is
|
|
absolute. A target escaping the root is refused rather than clamped: a
|
|
clamped path names a real file the bundle never pointed at.
|
|
"""
|
|
if target.startswith("/"):
|
|
return None
|
|
parts: list[str] = parent.split("/") if parent else []
|
|
for segment in target.split("/"):
|
|
if segment in ("", "."):
|
|
continue
|
|
if segment == "..":
|
|
if not parts:
|
|
return None
|
|
parts.pop()
|
|
continue
|
|
parts.append(segment)
|
|
return "/".join(parts) if parts else None
|
|
|
|
|
|
def bundle_ref(bundle_root: Path, *, profile: BundleProfile = DEFAULT_PROFILE) -> str:
|
|
"""A content identity for the bundle: `sha256-tree:<hex>`.
|
|
|
|
The digest is taken over the LF-joined, byte-sorted lines
|
|
`<posix-relative-path>\t<sha256 of the file bytes>` across every file the
|
|
INDEX TREE reaches -- the indexes themselves and the concepts they name.
|
|
|
|
**Reachable, not `rglob`, and the cost is stated rather than hidden.** A
|
|
file in the directory that no index names is outside this identity. It is
|
|
also outside what the pre-pass may read (SS 9.2), so an identity covering it
|
|
would be an assertion about bytes this command is forbidden to look at. The
|
|
property that matters holds: every byte that can reach a payload is inside
|
|
the ref, so the ref moves whenever a delivered excerpt could.
|
|
|
|
**Path and bytes, never mtime or size.** A digest over metadata would move
|
|
on a copy and hold still on an edit that preserved length, which is the
|
|
opposite of what an identity is for.
|
|
"""
|
|
indexes, concepts = _walk_index_tree(bundle_root, profile=profile)
|
|
suffix = profile.paths.concept_suffix
|
|
lines: list[bytes] = []
|
|
for relative in (*indexes, *(f"{concept}{suffix}" for concept in concepts)):
|
|
path = bundle_root / relative
|
|
if not path.is_file():
|
|
continue
|
|
digest = hashlib.sha256(path.read_bytes()).hexdigest()
|
|
lines.append(f"{relative}\t{digest}".encode())
|
|
joined = b"\n".join(sorted(lines))
|
|
return f"{REF_ALGORITHM}:{hashlib.sha256(joined).hexdigest()}"
|