#!/usr/bin/env python3 """Propose a segmentation plan for one document. A human adjudicates it. Pipeline step 3, and deliberately OUTSIDE the package. `src/` promises zero model calls on the run path, and the split of a document into units of knowledge is a judgement. Keeping the judgement lane out here is what lets the run path stay a deterministic replay of a decision somebody already made. ## What the research says this tool may and may not claim Topic 2 measured the OKF reference agent's granularity criteria against `_okf-canonical`: it splits on **what a thing is**, not on layout, and makes "multiple `write_concept_doc` calls ... rather than dumping everything into one doc". Four of its gates are semantic and need a model. A handful of MECHANICAL rules port today, and those are the ones below. Topic 1b measured heading derivation on the K2 corpus: 11 of 11 prose headings recovered -- from ONE document. 23 of 33 PDFs carry no outline at all and 95 % of the outline entries that do exist are AutoCAD export metadata. The denominator is 1. A rule validated on n=1 is not validated, and this tool says so by marking every entry it emits `PROPOSED` rather than adjudicated. Topic 1a measured that the best deterministic heading rule from poppler is a CONJUNCTION -- `size AND bold`, via `-fontfullname` -- at recall 1.000 and precision 0.846, and that adding weight as a DISJUNCT makes precision worse (0.786 -> 0.524). That path is implemented here and nowhere else: poppler is a SYSTEM binary the `[extract]` extra cannot express, so it may never be on the run path or in a golden fixture. ## The one rule that is not a heuristic **Nothing here is ever adjudicated.** `adjudicated: false` sits at the top of every artifact and `PROPOSED` in every entry's `derived` list. A plan is replayed deterministically and forever by the run path, so a proposal that could pass for an adjudication would put a machine's guess where a human's judgement is supposed to be, permanently and silently. Stdlib only. No network: the model-backed path this tool deliberately does not have would need the per-run network opt-in, and the socket-free test suite proves the absence rather than assuming it. """ from __future__ import annotations import argparse import hashlib import json import re import sys import unicodedata from dataclasses import dataclass from pathlib import Path from typing import Any sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) from llm_ingestion_okf.errors import IngestError # noqa: E402 from llm_ingestion_okf.extract import extract_text # noqa: E402 from llm_ingestion_okf.materialize import reduce_to_id_grammar # noqa: E402 #: Stamped into every entry's `derived` list. The marker is what keeps a #: proposal from being mistaken for the judgement the run path replays. PROPOSED_MARKER = "PROPOSED" #: This tool's identity, written into the artifact so an operator reading a #: plan six months later can tell what produced it. PROPOSER_ID = "okf-propose-segments" PROPOSER_VERSION = "1" #: The rules that survived Topic 2's port test. Each entry names exactly one, #: so a proposal an operator disagrees with is traceable to the rule that made #: it rather than to the tool as a whole. RULE_HEADING = "rule:heading" RULE_TABLE_BLOCK = "rule:table-block" RULE_POPPLER_SIZE_AND_BOLD = "rule:poppler-size-and-bold" RULE_NAMES = (RULE_HEADING, RULE_TABLE_BLOCK, RULE_POPPLER_SIZE_AND_BOLD) #: Norwegian and English function words. A heading made only of these names no #: unit of knowledge -- it is a connective that happened to sit on its own line. #: Topic 2's stop-word gate, and the only place this tool judges wording. STOP_WORDS = frozenset( { "and", "as", "at", "av", "be", "by", "da", "de", "den", "der", "det", "en", "er", "et", "for", "fra", "i", "in", "is", "it", "med", "of", "og", "om", "on", "or", "over", "paa", "som", "til", "the", "to", "under", "ved", "with", } ) # An ATX heading, or a numbered section opening a line (`3.1 Brannkonsept`). # A BARE integer is not a section number, for the same reason `structure.py` # refuses one: `12 ting` is an ordinary line and admitting it would cut a # document at every list item. _ATX = re.compile(r"^(?P#{1,6})\s+(?P\S.*?)\s*$") _NUMBERED = re.compile(r"^(?P<number>\d+(?:\.\d+)+)\s+(?P<title>\S.*?)\s*$") _TABLE_ROW = re.compile(r"^\s*\|.*\|\s*$") class ProposerError(Exception): """The run failed. NOT 'nothing to propose' -- the two must stay distinct.""" @dataclass(frozen=True) class Candidate: """One proposed boundary, before it becomes an entry.""" title: str level: int number: str | None rule: str start: int end: int def _is_stop_word_only(title: str) -> bool: words = [word for word in re.split(r"[^\w]+", title.lower()) if word] return bool(words) and all(word in STOP_WORDS for word in words) def find_candidates(text: str) -> list[Candidate]: """Every boundary the mechanical rules propose, in document order. Two gates from Topic 2 are applied here and both REMOVE candidates: - the **stop-word gate**: a heading made only of function words is not a unit of knowledge; - the **orphan check**: a heading with no body under it proposes nothing, because an empty concept is the silent skip this library refuses everywhere else. """ lines = text.splitlines(keepends=True) offsets: list[int] = [] position = 0 for line in lines: offsets.append(position) position += len(line) end_of_text = position marked: list[tuple[int, Candidate]] = [] in_table = False for index, line in enumerate(lines): if _TABLE_ROW.match(line): if not in_table: in_table = True marked.append( ( index, Candidate( title=f"Tabell linje {index + 1}", level=9, number=None, rule=RULE_TABLE_BLOCK, start=offsets[index], end=end_of_text, ), ) ) continue in_table = False atx = _ATX.match(line) numbered = _NUMBERED.match(line) if atx is None and numbered is None: continue if atx is not None: title = atx.group("title") level = len(atx.group("hashes")) inner = _NUMBERED.match(title) number = inner.group("number") if inner else None else: assert numbered is not None title = numbered.group("title") number = numbered.group("number") level = number.count(".") + 1 # The stop-word gate. Applied to the TITLE, after any section number # has been split off, so `3.1 Og` is judged on `Og`. if _is_stop_word_only(title): continue marked.append( ( index, Candidate( title=title, level=level, number=number, rule=RULE_HEADING, start=offsets[index], end=end_of_text, ), ) ) candidates: list[Candidate] = [] for position_in_list, (_, candidate) in enumerate(marked): following = marked[position_in_list + 1 :] end = offsets[following[0][0]] if following else end_of_text body = text[candidate.start : end] # The orphan check: everything after the heading line itself. if not body.splitlines()[1:] or not "".join(body.splitlines()[1:]).strip(): continue candidates.append( Candidate( title=candidate.title, level=candidate.level, number=candidate.number, rule=candidate.rule, start=candidate.start, end=end, ) ) return candidates def _segment_path(candidate: Candidate, taken: set[str]) -> str: title = unicodedata.normalize("NFC", candidate.title) # The section number becomes the DIRECTORY, so leaving it in the stem too # yields `3-1/3-1-brannkonsept.md` -- correct and unreadable. if candidate.number and title.startswith(candidate.number): title = title[len(candidate.number) :] stem = reduce_to_id_grammar(title) if not stem: stem = "seksjon" directory = reduce_to_id_grammar(candidate.number or "") if candidate.number else "" path = f"{directory}/{stem}.md" if directory else f"{stem}.md" suffix = 2 while path in taken: path = f"{directory}/{stem}-{suffix}.md" if directory else f"{stem}-{suffix}.md" suffix += 1 taken.add(path) return path def build_plan( source: Path, text: str, source_bytes: bytes, *, okf_type: str, proposed_at: str ) -> dict[str, Any]: """The artifact. Every entry PROPOSED, the plan itself never adjudicated.""" taken: set[str] = set() entries: list[dict[str, Any]] = [] for candidate in find_candidates(text): entries.append( { "segment_id": f"p{len(entries) + 1}", "path": _segment_path(candidate, taken), "title": candidate.title, "okf_type": okf_type, "span": [candidate.start, candidate.end], "ingested_at": proposed_at, # PROPOSED first, then the rule that proposed it. `derived` is # this library's existing "which of these did we infer" marker, # so a consumer that already distrusts derived fields # distrusts these by construction. "derived": [PROPOSED_MARKER, candidate.rule], } ) return { "version": "1", "source_sha256": hashlib.sha256(source_bytes).hexdigest(), "extractor_id": source.suffix.lower().lstrip(".") or "none", "extractor_version": PROPOSER_VERSION, "adjudicated_at": proposed_at, # NOT a timestamp question. `adjudicated_at` records when this artifact # was produced; this records whether a human has looked at it, and it is # false until one replaces the file. "adjudicated": False, "proposed_by": f"{PROPOSER_ID}/{PROPOSER_VERSION}", "entries": entries, } def run(source: Path, out: Path, *, okf_type: str, proposed_at: str) -> int: if not source.is_file(): raise ProposerError(f"source is not a file: {source}") try: source_bytes = source.read_bytes() except OSError as exc: raise ProposerError(f"cannot read {source}: {exc}") from exc try: text = extract_text(source.name, source_bytes) except IngestError as exc: raise ProposerError(f"cannot extract text from {source.name}: {exc}") from exc payload = build_plan(source, text, source_bytes, okf_type=okf_type, proposed_at=proposed_at) out.parent.mkdir(parents=True, exist_ok=True) out.write_bytes((json.dumps(payload, indent=2, ensure_ascii=False) + "\n").encode("utf-8")) print( f"{PROPOSER_ID}: proposed {len(payload['entries'])} segment(s) -> {out}\n" f"{PROPOSER_ID}: every entry is PROPOSED. Adjudicate before ingesting.", file=sys.stderr, ) return 0 def parse_args(argv: list[str] | None) -> argparse.Namespace: parser = argparse.ArgumentParser( prog=PROPOSER_ID, description="Propose a segmentation plan. A human adjudicates it before use.", ) parser.add_argument("source", type=Path, help="the document to segment") parser.add_argument("--out", type=Path, required=True, help="where to write the artifact") parser.add_argument("--okf-type", default="reference", help="okf_type for every entry") parser.add_argument( "--proposed-at", default="1970-01-01T00:00:00Z", help="the timestamp written into the artifact; explicit so a run is reproducible", ) return parser.parse_args(argv) def main(argv: list[str] | None = None) -> int: args = parse_args(argv) try: return run(args.source, args.out, okf_type=args.okf_type, proposed_at=args.proposed_at) except ProposerError as exc: print(f"{PROPOSER_ID}: FAILED - {exc}", file=sys.stderr) print( f"{PROPOSER_ID}: this is NOT 'nothing to propose'. Nothing was written.", file=sys.stderr, ) return 2 if __name__ == "__main__": raise SystemExit(main())