A PDF carries no notion of a heading -- a heading in a PDF is a typographic
fact -- so the text stream `pdfplumber` hands the segment proposer has already
thrown away the only evidence there was. The `docx` path never had that problem:
the converter emits ATX headings and `_ATX` cuts on them. Two readers close the
gap, and both are OFF.
`--pdf-headings font` infers a heading from the conjunction this repository
already measured (size above the document's character-weighted body median AND
a bold font name, recall 1.000 / precision 0.846) and emits it as ATX in the
SAME markdown the office path produces, so `_ATX` applies unchanged and no
PDF-only heading grammar exists.
It stays off BY MEASUREMENT, and the measurement is the point of the round:
against the operator's unit worksheet it takes `pdf` from 2 of 8 to 0 of 8,
losing two exact matches. The mechanism of the loss is stated rather than
guessed -- on those documents the outline rule already recovers the document's
own numbered chapters, so a second heading source can only add. Whole-corpus
screen: 25 of 32 `pdf` change, 0 of 5 `docx`, 0 of 2 `xlsx`. The default bundle
is byte-identical before and after this commit (`diff -r`, exit 0).
`--ocr` reads a page as an image when its own text never arrived: empty, or
`(cid:N)` placeholder codes at or above a threshold READ OFF a measured
distribution -- 834 pages over 32 files, 818 at exactly 0.0 and 16 at 0.93 or
above, nothing in between. On the one corpus document with the failure: 95.07 %
cid to 0 %, 44 to 2561 words of four or more letters, 17 to 18 pages with text.
Its engine is an optional dependency group and never a runtime dependency; a
packaging test pins both halves, and without the group every affected file is a
coded rejection (`extractor_ocr_group_missing`) rather than a crash.
Also corrects two stale published facts found while measuring: the README still
said two segmentation rules were on by default after `f6fea13` made it three,
and CLAUDE.md's K2 digest named the round-3 default. The current default is
492 concepts / 944 files, `bdefa679...`.
Report: docs/2026-09-08-k3-runde4-pdf-skrift-og-ocr.md
Co-Authored-By: Claude <claude-opus-5>
197 lines
10 KiB
TOML
197 lines
10 KiB
TOML
[build-system]
|
|
requires = ["hatchling"]
|
|
build-backend = "hatchling.build"
|
|
|
|
[project]
|
|
name = "llm-ingestion-okf"
|
|
version = "0.6.0"
|
|
description = "Shared OKF (Open Knowledge Format) ingestion library: spec-based connectors, bundle inbox, and external-bundle import, with security delegated to llm-ingestion-guard."
|
|
readme = "README.md"
|
|
license = "MIT"
|
|
requires-python = ">=3.10"
|
|
authors = [{ name = "Kjell Tore Guttormsen" }]
|
|
classifiers = [
|
|
"Development Status :: 3 - Alpha",
|
|
"Intended Audience :: Developers",
|
|
"Operating System :: OS Independent",
|
|
"Programming Language :: Python :: 3",
|
|
"Programming Language :: Python :: 3.10",
|
|
]
|
|
# Exactly one runtime dependency, ever: the security boundary. Everything
|
|
# else is stdlib. The version range is the real pin — it resolves normally
|
|
# against a package index, and is satisfied today by the git+https tag
|
|
# install documented in the README (a direct reference is an install-time
|
|
# channel, not a dependency declaration).
|
|
#
|
|
# Floor 1.2, not the 1.0.0 freeze: this library needs the flow-mapping
|
|
# frontmatter support (`generated: { by: x, at: y }`) that landed in the
|
|
# guard's 1.2.0, without which Door C fail-secures every concept carrying
|
|
# it. Ceiling <2.0, not a narrower minor: the guard's own 1.0.0 release
|
|
# promises no exported name is removed, renamed or given a different
|
|
# meaning short of a 2.0.0 — calibration (severities, dispositions) is
|
|
# explicitly free to move within 1.x under that same promise, so a tighter
|
|
# ceiling here would claim a stability guarantee the guard does not need to
|
|
# keep and we do not need to demand.
|
|
dependencies = ["llm-ingestion-guard>=1.2,<2.0"]
|
|
|
|
# The installed command. `okf build <folder> --bundle <dir>` is the packaged
|
|
# form of a path that was two unpackaged scripts under `tools/` and nine flags
|
|
# -- reachable only from a clone, which is not where a consumer stands.
|
|
[project.scripts]
|
|
okf = "llm_ingestion_okf.cli:main"
|
|
|
|
[project.optional-dependencies]
|
|
# Binary file-type extraction parsers. OPT-IN ONLY: this extra pulls binary
|
|
# wheels (pillow, pypdfium2) and a transitive tree that core must never have —
|
|
# the "exactly one runtime dependency" rule above covers the default install,
|
|
# and this extra is outside it by construction.
|
|
#
|
|
# The extra names the parsers it actually ships, so a consumer installing it
|
|
# gets what the error message promised and nothing else. It ships two: a `pdf`
|
|
# reader, and a converter that reaches the office types.
|
|
#
|
|
# WHY pdfplumber, and why the floor is not free (measured 2026-08-21,
|
|
# docs/2026-08-21-g2-pdf-extraction-measurement.md): on a real Vegnormalene
|
|
# requirement table pdfplumber keeps 4 of 4 rows with label and value on the
|
|
# same line; pypdf, pdfminer.six and pymupdf each keep 0 of 4, emitting all
|
|
# labels then all values, which a downstream reader can only re-pair by
|
|
# guessing. In a `krav` document that is a wrong answer that looks right.
|
|
# pymupdf is additionally out on LICENSE (AGPL-3.0 or commercial) — this
|
|
# package is MIT and an extra must not hand a consumer copyleft they did not
|
|
# choose.
|
|
#
|
|
# PARSER VERSION IS PART OF THE OUTPUT CONTRACT. pdfplumber pins
|
|
# `pdfminer.six==20260107` exactly, and pdfminer.six ships date-stamped
|
|
# releases with no stability contract. Extraction is deterministic WITHIN a
|
|
# parser version (measured, 5 configurations) and NOT guaranteed across one.
|
|
# `tests/test_extract.py` holds that promise against a committed fixture, so
|
|
# widening this range makes a test go red instead of letting extracted text
|
|
# drift silently. See tests/fixtures/README.md.
|
|
#
|
|
# WHY THE CONVERTER BINARY IS VENDORED RATHER THAN FOUND ON PATH. The `xlsx`
|
|
# and `pptx` readers exist only from pandoc 3.8.3. Debian 12 ships 2.17.1.1
|
|
# and Ubuntu 24.04 ships 3.1.3, so a PATH binary cannot deliver two of the
|
|
# five office formats on current stable distributions -- and a library whose
|
|
# output depends on which pandoc a host happens to carry is not deterministic
|
|
# in the sense the rest of this package means it.
|
|
#
|
|
# `pypandoc-binary` carries the binary inside the wheel (7 platform wheels at
|
|
# 1.17, including macosx x86_64/arm64, manylinux and musllinux x86_64/aarch64,
|
|
# and win_amd64 -- measured on the PyPI JSON API 2026-09-02). The pin is
|
|
# EXACT, not a range, because the binary's version is part of the output
|
|
# contract in the same way pdfminer.six's is: extraction is deterministic
|
|
# within a converter version and not across one.
|
|
#
|
|
# This does not widen the runtime dependency surface. The rule above governs
|
|
# `project.dependencies`, which still names the guard alone; the extra is
|
|
# outside it by construction, and the test below now pins its contents so a
|
|
# third entry cannot arrive unexamined.
|
|
extract = ["pdfplumber>=0.11.10,<0.12", "pypandoc-binary==1.17"]
|
|
|
|
# The OCR engine for `--ocr`, and NEVER a runtime dependency. It is a separate
|
|
# group from `extract` rather than three more entries in it, because it buys
|
|
# something categorically different: `extract` decides which file types can be
|
|
# read at all, while this one only changes how a PDF page is read when the
|
|
# page's own text never arrived. A consumer who installs `[extract]` gets every
|
|
# file type; a consumer who never meets a scanned document should never carry
|
|
# an inference runtime.
|
|
#
|
|
# WHY rapidocr ON onnxruntime, and why not the obvious alternative. Docling was
|
|
# measured first and is OUT on a platform fact, not a preference: it needs
|
|
# torch, and torch stopped publishing macOS x86_64 wheels after 2.2.2, with no
|
|
# `transformers` version inside Docling's own window that works against that
|
|
# one (4 tried, 2026-09-08). rapidocr on onnxruntime installs and runs on this
|
|
# machine, and it carries its ONNX models inside its own wheel, so `--ocr`
|
|
# needs no network at run time -- which matters here more than usual, since
|
|
# this library's network gate is an explicit per-run opt-in and an engine that
|
|
# downloaded a model on first use would walk straight through it.
|
|
#
|
|
# `pypdfium2` is named although `[extract]` already reaches it through
|
|
# pdfplumber: the OCR path RENDERS a page before reading it, and the renderer
|
|
# is a dependency of that path rather than a happy accident of another one.
|
|
#
|
|
# The pins are ranges rather than exact versions, and that is a weaker promise
|
|
# than `[extract]` makes on purpose: OCR output is a model's reading of an
|
|
# image, so it is deterministic within one model version and NOT across one,
|
|
# and no range can make it otherwise. A bundle built with `--ocr` is
|
|
# reproducible against the versions it was built with, which is stated in the
|
|
# report rather than implied by a pin.
|
|
ocr = ["rapidocr>=3.9,<4", "onnxruntime>=1.20,<2", "pypdfium2>=4,<6"]
|
|
|
|
[dependency-groups]
|
|
dev = ["pytest>=8", "mypy>=1.14", "ruff>=0.9"]
|
|
|
|
[tool.hatch.build.targets.wheel]
|
|
packages = ["src/llm_ingestion_okf"]
|
|
|
|
# Two AUTHORED files the packaged commands cannot run without, carried into the
|
|
# wheel from where they are edited rather than committed a second time under
|
|
# `src/`. A duplicate would drift, and both of these are checked against
|
|
# literals in the code: a template whose blocks are pinned by a test, and a
|
|
# known-positive artefact whose byte count is a constant in `consume.py`.
|
|
#
|
|
# `okf skill` instantiates the template; `okf consume` measures the contract
|
|
# document as its section 7.4 known-positive and refuses without it. Before
|
|
# 2026-09-08 neither command was installable, so neither file had to travel.
|
|
[tool.hatch.build.targets.wheel.force-include]
|
|
"skills/okf-consume-template/SKILL.md" = "llm_ingestion_okf/_data/okf-consume-template.md"
|
|
"docs/consumption-contract.md" = "llm_ingestion_okf/_data/consumption-contract.md"
|
|
|
|
[tool.ruff]
|
|
line-length = 100
|
|
target-version = "py310"
|
|
|
|
[tool.mypy]
|
|
strict = true
|
|
python_version = "3.10"
|
|
|
|
# llm-ingestion-guard ships no py.typed marker, so its symbols arrive as Any.
|
|
# The adapter coerces every value it carries across the seam to a concrete
|
|
# type, which is what keeps --strict meaningful on this side of it.
|
|
[[tool.mypy.overrides]]
|
|
module = ["llm_ingestion_guard", "llm_ingestion_guard.*"]
|
|
ignore_missing_imports = true
|
|
|
|
# `pypandoc` ships no py.typed marker either. Only `_pandoc.py` imports it, and
|
|
# every value it hands back is coerced to `str`/`Path` there before it reaches
|
|
# the rest of the package -- the same discipline as the guard adapter above.
|
|
[[tool.mypy.overrides]]
|
|
module = ["pypandoc", "pypandoc.*"]
|
|
ignore_missing_imports = true
|
|
|
|
# `rapidocr` ships no py.typed marker either, and it is behind an OPTIONAL
|
|
# group -- so on a machine without that group installed the import does not
|
|
# resolve at all. Only `_ocr_reader` imports it, and the only value that
|
|
# crosses back is coerced to `str` there, the same discipline as the two
|
|
# overrides above.
|
|
[[tool.mypy.overrides]]
|
|
module = ["rapidocr", "rapidocr.*"]
|
|
ignore_missing_imports = true
|
|
|
|
# Install CHANNEL for the guard, which is not on a package index yet. It is
|
|
# uv-specific, and it reaches further than a dev-only setting: a consumer
|
|
# installing this package from git WITH UV picks the guard up from this tag
|
|
# automatically, because uv reads this file when it builds from the source
|
|
# tree. Measured against an empty cache 2026-07-25, 2026-08-20, and
|
|
# 2026-08-21 on uv 0.9.8. The 08-21 run also measured the TRANSITIVE form: a
|
|
# separate consumer project naming only this package still resolves the guard
|
|
# from the entry below, because this package reaches it as a git source.
|
|
#
|
|
# That source is the whole reach. A wheel carries Requires-Dist and nothing
|
|
# else, so this entry cannot survive an index install — and while the guard is
|
|
# off-index, removing it would break the one-command uv path the README
|
|
# documents.
|
|
#
|
|
# pip does not read it at all: it resolves [project.dependencies] alone and
|
|
# fails with "No matching distribution found for llm-ingestion-guard" until
|
|
# the guard is installed from its own tag first (README; measured 2026-08-21,
|
|
# both the failure and the two-command recovery).
|
|
#
|
|
# Either way the range above stays the pin, and the pin is per-tree: a wheel
|
|
# built from THIS tree carries `Requires-Dist: llm-ingestion-guard<2.0,>=1.2`,
|
|
# measured 2026-08-23 against the built wheel. The `<0.4,>=0.3` this comment
|
|
# carried before was the `v0.3.4` tag's range — still true of that tag, never
|
|
# true of this tree. Reading a range off one and installing it against the
|
|
# other is the one combination that fails.
|
|
[tool.uv.sources]
|
|
llm-ingestion-guard = { git = "https://git.fromaitochitta.com/open/llm-ingestion-pipeline-security.git", tag = "v1.3.0" }
|