1
0
Fork 0

docs(okf): measure the tags/description gap against the pinned SPEC and corpus

Order 20260906T213322Z: measure and propose, no src, no bump. Measured against
SPEC _okf-canonical @ ad30107 and the corpus _okf-upstream @ 3fcbb9f (denominator
53, extracted at the pin because the work tree had moved on to 9a15b13).

Three claims in the tags/description entry are stale against the 1.3.0 parser.
The parser does have a sequence type -- _consume_block_list parses block lists of
scalars -- what it lacks is tolerance for the flush-left indentation 36/53 of the
corpus uses. Removing tags alone now lets 6/53 pass, not the 4/53 recorded, and
the entry's headline claim does not hold at all: description alone unblocks 0/53,
and with all three surface forms closed 44/53 still stop on generated as a
top-level block mapping. The entry oversells its own reach by a factor of seven.

The SPEC has no depth rule. no-nesting-past-depth-1 is entirely ours; the
conformance floor is only "a parseable YAML frontmatter block" (SS11.1).

Candidates were measured by normalizing the surface form onto a shape the parser
already accepts, then importing parse_frontmatter -- the predicate is never
re-implemented, only the input's spelling is rewritten. P1 (scalar flow sequence)
takes the corpus from 0/53 to 6/53 without spending depth-1; P2 and P3 buy 0/53
each and 6/53 stacked on P1. Recommendation is P1 alone, and the note says out
loud that this does not open the corpus.

The label "punkt 44" is not grep-able: it came from a count-after-insertion, and
the entry is today the 12th of 45 at docs/LIMITATIONS.md:126.

No src change, no version bump, no push. 868 passed, coverage 130/130 + 6/6 gaps,
redos sweep exit 0 -- all after git add.
This commit is contained in:
Kjell Tore Guttormsen 2026-09-06 23:58:02 +02:00
commit 6e7c8d2b98
5 changed files with 579 additions and 0 deletions

155
scratchpad/candidates.py Normal file
View file

@ -0,0 +1,155 @@
"""What each candidate predicate would admit, measured by normalizing the SURFACE
form onto a shape the CURRENT parser already accepts. The predicate is imported,
never re-implemented: only the input's spelling is rewritten."""
import os, re, sys, collections
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "src"))
from llm_ingestion_guard import okf
ROOT = os.path.join(os.path.dirname(__file__), "corpus", "okf", "bundles")
def docs():
out = []
for dirpath, _d, fns in os.walk(ROOT):
for fn in sorted(fns):
if fn.endswith(".md") and fn not in ("index.md", "log.md"):
out.append(os.path.join(dirpath, fn))
return sorted(out)
def split_fm(text):
lines = text.split("\n")
if not lines or lines[0].strip() != "---":
return None, None
for i in range(1, len(lines)):
if lines[i].strip() == "---":
return lines[1:i], lines[i + 1:]
return None, None
TOPKEY = re.compile(r"^([A-Za-z0-9_]+):(.*)$")
def n_flow_scalar_seq(fm):
"""P1: `k: [a, b]` where every element is a plain scalar -> indented block list."""
out = []
for raw in fm:
m = TOPKEY.match(raw)
if m and m.group(2).strip().startswith("[") and m.group(2).strip().endswith("]"):
inner = m.group(2).strip()[1:-1]
elems = [e.strip() for e in inner.split(",")]
if inner and all(e and "{" not in e and "}" not in e and "[" not in e for e in elems):
out.append("%s:" % m.group(1))
out.extend(" - %s" % e for e in elems)
continue
out.append(raw)
return out
def n_flush_block_seq(fm):
"""P2: a flush-left `- item` run under a bare `k:` -> the same run, indented."""
out, i, n = [], 0, len(fm)
while i < n:
raw = fm[i]
m = TOPKEY.match(raw)
if m and m.group(2).strip() == "" and i + 1 < n and fm[i + 1][:1] == "-":
out.append(raw)
i += 1
while i < n and (fm[i][:1] == "-" or fm[i][:1] in (" ", "\t")):
out.append(" " + fm[i])
i += 1
continue
out.append(raw)
i += 1
return out
def n_folded_scalar(fm):
"""P3: `k: text` continued on more-indented plain lines -> one joined line."""
out, i, n = [], 0, len(fm)
while i < n:
raw = fm[i]
m = TOPKEY.match(raw)
if m and m.group(2).strip() and not m.group(2).strip()[0] in "[{":
acc = raw
i += 1
while i < n and fm[i][:1] in (" ", "\t") and not fm[i].strip().startswith("- "):
acc = acc.rstrip() + " " + fm[i].strip()
i += 1
out.append(acc)
continue
out.append(raw)
i += 1
return out
def parses(fm, body):
text = "---\n" + "\n".join(fm) + "\n---\n" + "\n".join(body)
try:
okf.parse_frontmatter(text)
return None
except okf.OKFFrontmatterError as exc:
return str(exc)
COMBOS = [
("baseline (v1.3.0 as shipped)", []),
("P1 scalar flow sequence", [n_flow_scalar_seq]),
("P2 flush block sequence", [n_flush_block_seq]),
("P3 folded plain scalar", [n_folded_scalar]),
("P1+P3", [n_flow_scalar_seq, n_folded_scalar]),
("P2+P3", [n_flush_block_seq, n_folded_scalar]),
("P1+P2", [n_flow_scalar_seq, n_flush_block_seq]),
("P1+P2+P3", [n_flow_scalar_seq, n_flush_block_seq, n_folded_scalar]),
]
def main():
paths = docs()
print("denominator: %d documents (pin 3fcbb9f, parser HEAD)" % len(paths))
for name, fns in COMBOS:
ok = 0
residual = collections.Counter()
for p in paths:
fm, body = split_fm(open(p, encoding="utf-8").read())
for fn in fns:
fm = fn(fm)
err = parses(fm, body)
if err is None:
ok += 1
else:
residual[err.split(":")[0]] += 1
top = "; ".join("%dx %s" % (c, r) for r, c in residual.most_common(3))
print(" %-30s %2d/%d residual: %s" % (name, ok, len(paths), top or "-"))
main()
# --- diagnostic only (NOT a proposal): size the real binding constraint ---
def n_block_mapping(fm):
"""D4: a top-level `k:` followed by indented `a: b` lines -> one flow mapping."""
out, i, n = [], 0, len(fm)
while i < n:
raw = fm[i]
m = TOPKEY.match(raw)
if m and m.group(2).strip() == "" and i + 1 < n and fm[i + 1][:1] in (" ", "\t") \
and not fm[i + 1].strip().startswith("- "):
pairs = []
i += 1
while i < n and fm[i][:1] in (" ", "\t") and not fm[i].strip().startswith("- "):
pairs.append(fm[i].strip())
i += 1
out.append("%s: { %s }" % (m.group(1), ", ".join(pairs)))
continue
out.append(raw)
i += 1
return out
print("\n--- diagnostic: what caps the corpus ABOVE the tags/description gap ---")
for name, fns in [
("D4 top-level block mapping", [n_block_mapping]),
("P1+P2+P3+D4", [n_flow_scalar_seq, n_flush_block_seq, n_folded_scalar, n_block_mapping]),
]:
ok = 0
residual = collections.Counter()
for p in docs():
fm, body = split_fm(open(p, encoding="utf-8").read())
for fn in fns:
fm = fn(fm)
err = parses(fm, body)
if err is None:
ok += 1
else:
residual[err.split(":")[0]] += 1
top = "; ".join("%dx %s" % (c, r) for r, c in residual.most_common(3))
print(" %-30s %2d/53 residual: %s" % (name, ok, top or "-"))

45
scratchpad/measure.py Normal file
View file

@ -0,0 +1,45 @@
"""Measure the pinned OKF corpus against the CURRENT parser. No src/ changes."""
import os, sys, collections
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "src"))
from llm_ingestion_guard import okf
ROOT = os.path.join(os.path.dirname(__file__), "corpus", "okf", "bundles")
def docs():
out = []
for dirpath, _dirnames, filenames in os.walk(ROOT):
for fn in sorted(filenames):
if not fn.endswith(".md"):
continue
if fn in ("index.md", "log.md"):
continue
out.append(os.path.join(dirpath, fn))
return sorted(out)
def main():
paths = docs()
ok, fail = [], []
reasons = collections.Counter()
for p in paths:
with open(p, encoding="utf-8") as fh:
text = fh.read()
try:
okf.parse_frontmatter(text)
ok.append(p)
except okf.OKFFrontmatterError as exc:
fail.append((p, str(exc)))
reasons[str(exc).split(":")[0]] += 1
print("denominator: %d documents" % len(paths))
print("parse OK : %d" % len(ok))
print("parse FAIL : %d" % len(fail))
print("--- failure classes ---")
for r, c in reasons.most_common():
print("%4d %s" % (c, r))
print("--- passing documents (positive control) ---")
for p in ok:
print(" " + os.path.relpath(p, ROOT))
print("--- first 8 failures verbatim ---")
for p, e in fail[:8]:
print(" %s\n %s" % (os.path.relpath(p, ROOT), e))
main()

59
scratchpad/shapes.py Normal file
View file

@ -0,0 +1,59 @@
"""Shape census of the pinned OKF corpus: which surface form each key is written in."""
import os, re, sys, collections
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "src"))
from llm_ingestion_guard import okf
ROOT = os.path.join(os.path.dirname(__file__), "corpus", "okf", "bundles")
def docs():
out = []
for dirpath, _d, fns in os.walk(ROOT):
for fn in sorted(fns):
if fn.endswith(".md") and fn not in ("index.md", "log.md"):
out.append(os.path.join(dirpath, fn))
return sorted(out)
def split_fm(text):
lines = text.split("\n")
if not lines or lines[0].strip() != "---":
return None, None
for i in range(1, len(lines)):
if lines[i].strip() == "---":
return lines[1:i], lines[i + 1:]
return None, None
def key_shape(fm, key):
for i, raw in enumerate(fm):
m = re.match(r"^(%s):(.*)$" % re.escape(key), raw)
if not m:
continue
val = m.group(2).strip()
follow = fm[i + 1] if i + 1 < len(fm) else ""
if val.startswith("["):
return "flow-sequence"
if val.startswith("{"):
return "flow-mapping"
if val:
if follow[:1] in (" ", "\t") and not follow.strip().startswith("- "):
return "scalar+continuation"
return "single-line-scalar"
if follow.strip().startswith("- "):
return "block-seq-indented" if follow[:1] in (" ", "\t") else "block-seq-flush"
if follow[:1] in (" ", "\t"):
return "block-mapping"
return "empty"
return "ABSENT"
def main():
paths = docs()
print("denominator: %d" % len(paths))
for key in ("tags", "description", "generated", "verified", "sources"):
c = collections.Counter()
for p in paths:
fm, _ = split_fm(open(p, encoding="utf-8").read())
c[key_shape(fm, key)] += 1
print("\n%s:" % key)
for shape, n in c.most_common():
print(" %-22s %d/%d" % (shape, n, len(paths)))
main()