"""Run a corpus through the whole path and report numbers, never a claim. The instrument behind K1 and K2. It exists because the alternative -- a person running the door by hand and writing down what they saw -- has already produced a number that was right about a directory that no longer existed. Three rules it enforces rather than describes: **K1b is a command.** The conservation identity `merged + Sigma(coded rejections) == N` is CHECKED here, and a run where it does not hold EXITS NON-ZERO. Asserted in prose it would be something a reader has to trust; as an exit status it fails the run that produced it. When it fails, the unaccounted files are NAMED -- "some file went missing" is not actionable. **`N` is computed, never typed.** It is the file count of the corpus directory, read at run time. A literal would keep passing after the corpus changed and would then report a fact about a directory that no longer exists. **Three counts, never one.** The guard sits between extraction and persist, so a healthy persisted count can hide a pile of quarantines. Extracted, gated and persisted are separate numbers for that reason. **The degenerate-merge rule is a DEFINITION, not a threshold: a merge is degenerate when the extracted text is zero characters after stripping whitespace.** A concept with an empty body cannot carry one unit of knowledge, so counting it as a merge would report extraction failure as success. The resolved converter path and version are printed in the output, because the vendored binary is bypassed silently otherwise -- measured three times, wheel 3.9 against host 3.10.2. """ from __future__ import annotations import argparse import sys import time from dataclasses import dataclass, replace from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) from llm_ingestion_okf.errors import IngestError # noqa: E402 from llm_ingestion_okf.extract import extract_text # noqa: E402 from llm_ingestion_okf.inbox import GateDecision, InboxResult, process_inbox # noqa: E402 from llm_ingestion_okf.profiles import STRUCTURED_V1 # noqa: E402 __all__ = [ "CorpusReport", "converter_identity", "is_degenerate", "main", "measure", "replace", "unaccounted_names", ] HARNESS_ID = "okf-corpus-run" def is_degenerate(text: str) -> bool: """Zero characters after stripping whitespace. The whole rule, in one line. A definition rather than a threshold on purpose: a threshold invites a later argument about where it should sit, and every such argument has to be had again the next time the corpus changes. """ return not text.strip() def converter_identity() -> tuple[str, str]: """The converter this run would use, resolved by path, and its version. Reported rather than assumed. `pypandoc` prefers the HIGHEST version it can find over the one this package vendored, so a run that did not say which binary produced its text would be unattributable. """ from llm_ingestion_okf._pandoc import PANDOC_VERSION, resolve_pandoc try: return (str(resolve_pandoc()), PANDOC_VERSION) except IngestError as exc: return (f"unresolved ({exc.code})", PANDOC_VERSION) def unaccounted_names( *, dropped: tuple[str, ...], merged: tuple[str, ...], coded: tuple[str, ...] ) -> tuple[str, ...]: """Every dropped file that is in neither column, in sorted order. The conservation check, isolated so it can be driven with an inventory the door could not produce. A harness whose failure path is unreachable is a harness that proves nothing when it passes. """ return tuple(sorted(set(dropped) - set(merged) - set(coded))) @dataclass(frozen=True) class CorpusReport: """One corpus run's numbers, every one of them with its denominator.""" corpus: str n: int extracted: int gated: int persisted: int substantive: int degenerate: int rejected: int seconds_total: float converter_path: str converter_version: str codes: tuple[tuple[str, int], ...] unaccounted: tuple[str, ...] @property def merged(self) -> int: return self.substantive + self.degenerate def render(self) -> str: per_file = self.seconds_total / self.n if self.n else 0.0 lines = [ f"# Corpus run: {self.corpus}", "", f"N (denominator, the directory's file count) = {self.n}", "", "## Three counts, never one", "", "The guard sits between extraction and persist, so a healthy persisted", "count can hide a pile of quarantines.", "", f"- extracted: {self.extracted}/{self.n}", f"- gated: {self.gated}/{self.n}", f"- persisted: {self.persisted}/{self.n}", "", "## The numerator, split", "", "A merge is degenerate when the extracted text is zero characters after", "stripping whitespace -- a definition, not a threshold.", "", f"- substantive: {self.substantive}/{self.n}", f"- degenerate: {self.degenerate}/{self.n}", f"- rejected (coded): {self.rejected}/{self.n}", "", f"merged + coded rejections = {self.merged + self.rejected}; N = {self.n}", "", "## Converter", "", f"- resolved converter path: {self.converter_path}", f"- pinned converter version: {self.converter_version}", "", "## Wall time", "", f"- total: {self.seconds_total:.2f} s", f"- per file: {per_file:.3f} s", "", "## Rejection codes", "", ] lines.extend( f"- `{code}`: {count}/{self.n}" for code, count in self.codes or (("(none)", 0),) ) if self.unaccounted: lines += ["", "## UNACCOUNTED", ""] lines.extend(f"- {name}" for name in self.unaccounted) return "\n".join(lines) + "\n" def _gate(text: str) -> GateDecision: return GateDecision(sanitized_text=text, disposition="warn") def _split_merges(corpus: Path, result: InboxResult) -> tuple[int, int]: """Merged files split into substantive and degenerate, by the stated rule. Re-extracted here rather than read back off the bundle: the rule is about the EXTRACTED text, and a concept body has already been through the gate. """ substantive = 0 degenerate = 0 for item in result.persisted: source = corpus / item.source_file try: text = extract_text(source.name, source.read_bytes()) except (IngestError, OSError): continue if is_degenerate(text): degenerate += 1 else: substantive += 1 return (substantive, degenerate) def measure(corpus: Path, bundle: Path, *, ingested_at: str) -> CorpusReport: """Run the corpus through the door and count what happened.""" dropped = tuple(sorted(path.name for path in corpus.iterdir() if path.is_file())) started = time.monotonic() result = process_inbox( corpus, bundle, ingested_at, okf_type="reference", gate=_gate, profile=STRUCTURED_V1, ) elapsed = time.monotonic() - started merged_names = tuple(item.source_file for item in result.persisted) blocked = result.quarantined + result.rejected coded_names = tuple(item.source_file for item in result.failed) + tuple( item.source_file for item in blocked ) counts: dict[str, int] = {} for failure in result.failed: counts[failure.error.code] = counts.get(failure.error.code, 0) + 1 for item in blocked: counts[item.disposition] = counts.get(item.disposition, 0) + 1 substantive, degenerate = _split_merges(corpus, result) path, version = converter_identity() return CorpusReport( corpus=str(corpus), n=len(dropped), # A file that reached the gate was extracted; the gate here persists # everything it sees, so the two differ only when a gate refuses. extracted=len(merged_names) + len(blocked), gated=len(merged_names) + len(blocked), persisted=len(merged_names), substantive=substantive, degenerate=degenerate, rejected=len(coded_names), seconds_total=elapsed, converter_path=path, converter_version=version, codes=tuple(sorted(counts.items())), unaccounted=unaccounted_names(dropped=dropped, merged=merged_names, coded=coded_names), ) def parse_args(argv: list[str] | None) -> argparse.Namespace: parser = argparse.ArgumentParser( description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter ) parser.add_argument("--corpus", type=Path, required=True, help="the directory to run") parser.add_argument("--report", type=Path, required=True, help="where to write the report") parser.add_argument("--bundle", type=Path, default=None, help="where to build the bundle") parser.add_argument( "--ingested-at", default="2026-09-02T00:00:00Z", help="stamped verbatim, as everywhere" ) return parser.parse_args(argv) def main(argv: list[str] | None = None) -> int: args = parse_args(argv) if not args.corpus.is_dir(): print(f"{HARNESS_ID}: FAILED - no corpus directory at {args.corpus}", file=sys.stderr) return 2 bundle = args.bundle or args.report.parent / f"{args.corpus.name}-bundle" report = measure(args.corpus, bundle, ingested_at=args.ingested_at) args.report.parent.mkdir(parents=True, exist_ok=True) args.report.write_text(report.render(), encoding="utf-8", newline="") print(report.render()) if report.unaccounted or report.merged + report.rejected != report.n: print( f"{HARNESS_ID}: K1b FAILED - merged ({report.merged}) + coded rejections " f"({report.rejected}) != N ({report.n}). Unaccounted: " f"{', '.join(report.unaccounted) or '(none named)'}", file=sys.stderr, ) return 1 return 0 if __name__ == "__main__": raise SystemExit(main())