feat(consume): the ranking can read a body without the door's link line

One parameter, `link_in_signal`, default `True` -- so no payload moves. `False`
scores the body with `inbox._link_enclosing`'s line removed, and removes it from
the stem vocabulary too, because `searchable_text` counts `df` over the text a
hit is scored on: taking the line out of one and leaving it in the other would
measure two different texts.

The recognition is the door's own constant plus its own place -- last in the
body, after a blank line, in the door's link form. The excerpt keeps the line
either way, so two readings of ONE bundle differ in ORDER alone, which is what
separates a ranking movement from a budget displacement.

Not a CLI flag: whether this becomes a default is K3-23's measurement, not an
implementation's decision.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Kjell Tore Guttormsen 2026-09-12 17:25:00 +02:00
commit 8e82da4682

View file

@ -59,6 +59,7 @@ from .inbox import (
ADJUDICATION_ADJUDICATED,
ADJUDICATION_PROPOSED,
ADJUDICATION_STATES,
ENCLOSING_SECTION,
)
from .materialize import parse_frontmatter
from .profiles import (
@ -934,16 +935,50 @@ def question_uses_cost_vocabulary(question: str) -> bool:
return any(in_cost_vocabulary(token) for token in normalise(question))
def searchable_text(concepts: Sequence["Concept"]) -> list[str]:
#: The DOOR's line, exactly as `inbox._link_enclosing` writes it: the two SS 6.1
#: words, a markdown link, and a bundle-absolute target. Built from the door's
#: own constant, never from prose, so a rename there breaks the recognition
#: instead of quietly loosening it.
_LINK_LINE = re.compile(rf"^{re.escape(ENCLOSING_SECTION)}: \[[^\]]*\]\(/[^)]*\)$")
def body_without_link_line(body: str) -> str:
"""`body` minus the one line the door appends to a heading-only body.
THE FILE KEEPS THE LINE AND SO DOES THE READER. This is what a caller uses
to score a body the way it read before `--shell-parent` existed, on the same
bytes, in the same process -- never a second build compared against a first.
Three conditions, all the door's own, and each one is a known-negative that
would otherwise be stripped: the line is LAST, a blank line stands before
it, and it matches the door's form (`_LINK_LINE`). A human sentence opening
with the same two words has no link and survives; the door's exact form with
prose under it is not last and survives. What cannot be told apart is a
human line that is byte-for-byte the door's own, last in a body -- an
ambiguity of the form itself, stated rather than hidden.
"""
lines = body.rstrip("\n").split("\n")
if len(lines) < 3 or lines[-2].strip() or not _LINK_LINE.match(lines[-1]):
return body
return "\n".join(lines[:-2]) + "\n"
def searchable_text(concepts: Sequence["Concept"], *, link_in_signal: bool = True) -> list[str]:
"""The text a concept is scored against, one string per concept.
The same two fields the ranker's two lexical signals read -- title plus
id, and body -- joined, so a `df` counted here is a `df` over exactly what
a hit can be scored on. Counting rarity over one field and matching on
another would weight a token by how rare it is somewhere it is not read.
`link_in_signal=False` reads the bodies without the door's link line, for
the same reason: the stem vocabulary and the rarity `df` are counted over
what a hit is scored on, so an instrument that took the line out of the
signal and left it in the vocabulary would measure two different texts.
"""
return [
f"{concept.title} {concept.concept_id.replace('/', ' ')} {concept.body}"
f"{concept.title} {concept.concept_id.replace('/', ' ')} "
f"{concept.body if link_in_signal else body_without_link_line(concept.body)}"
for concept in concepts
]
@ -1368,6 +1403,7 @@ def concept_scores(
tie_shared_rank: bool = DEFAULT_TIE_SHARED_RANK,
title_covered: bool = DEFAULT_TITLE_COVERED,
stems: frozenset[str] | None = None,
link_in_signal: bool = True,
) -> list[tuple[Concept, float, int]]:
"""Every concept, ordered best first, fused from three signals by RRF.
@ -1412,6 +1448,17 @@ def concept_scores(
only `tunnel` and `vann` whose ids sorted earlier. With shared ranks it
fuses to rank 3.
**`link_in_signal=False` is an INSTRUMENT, not a flag and not a default.**
The body this reads is the file's, and `inbox._link_enclosing` puts one line
in it -- `Enclosing section: [<title>](/<bundle-absolute path>)` -- whose
path repeats the document's own directory in every linked body. SS 6.1's
argument for the absolute form is about the FILE; nothing decided it should
be SCORED. Passing `False` scores the body without that line, on the same
bytes and in the same process, so the two readings can be measured against
each other rather than two builds. The excerpt a reader gets is untouched
either way; no caller in the run path passes it and the CLI does not expose
it.
The rule takes the FIRST position of a score group rather than its middle.
Both were measured on the same four cases; the middle put the same concept
at rank 5 where the first puts it at 3, and neither changed the three
@ -1425,6 +1472,12 @@ def concept_scores(
# The id BELOW the directories every concept shares: a segment every
# concept carries separates nothing. See `shared_id_prefix`.
shared = shared_id_prefix([concept.concept_id for concept in concepts])
bodies = {
concept.concept_id: (
concept.body if link_in_signal else body_without_link_line(concept.body)
)
for concept in concepts
}
titles = {
concept.concept_id: f"{concept.title} {' '.join(concept.concept_id.split('/')[shared:])}"
for concept in concepts
@ -1446,7 +1499,7 @@ def concept_scores(
concept.concept_id: float(
_overlap(
question_tokens,
concept.body,
bodies[concept.concept_id],
cost_vocabulary=bridge,
weights=weights,
stems=stems,
@ -1498,7 +1551,12 @@ def concept_scores(
_overlap(
question_tokens, titles[concept.concept_id], cost_vocabulary=bridge, stems=stems
)
+ _overlap(question_tokens, concept.body, cost_vocabulary=bridge, stems=stems)
+ _overlap(
question_tokens,
bodies[concept.concept_id],
cost_vocabulary=bridge,
stems=stems,
)
)
for concept in concepts
}
@ -1966,6 +2024,7 @@ def build_payload(
stem_prefix: bool = DEFAULT_STEM_PREFIX,
source_quota: int | None = DEFAULT_SOURCE_QUOTA,
follow_parent: bool = DEFAULT_FOLLOW_PARENT,
link_in_signal: bool = True,
) -> dict[str, object]:
"""One bundle plus one question, cut to one contract-conformant payload.
@ -1989,6 +2048,13 @@ def build_payload(
would move every consumer's bytes for a field none of them asked for.
Whether the naming is worth the bookkeeping is the caller's call, and the
flag is how it stays one.
**`link_in_signal` is an instrument and carries no CLI flag** -- see
:func:`concept_scores`. `False` keeps the door's `Enclosing section:` line
out of the ranking and out of the stem vocabulary; the excerpt still carries
it, so a payload built either way holds the same bytes per excerpt and only
the ORDER can differ. That is what separates a ranking movement from a
budget displacement on one bundle.
"""
case, expected, measured = known_positive()
if expected != measured:
@ -2019,12 +2085,20 @@ def build_payload(
# separates a Norwegian compound from four coincidental characters. One
# pass, over the same text the ranking reads.
stems = (
frozenset(token for text in searchable_text(concepts) for token in normalise(text))
frozenset(
token
for text in searchable_text(concepts, link_in_signal=link_in_signal)
for token in normalise(text)
)
if stem_prefix
else None
)
weights = (
rarity_weights(normalise(question), searchable_text(concepts), stems=stems)
rarity_weights(
normalise(question),
searchable_text(concepts, link_in_signal=link_in_signal),
stems=stems,
)
if rarity_weight
else None
)
@ -2044,6 +2118,7 @@ def build_payload(
tie_shared_rank=tie_shared_rank,
title_covered=title_covered,
stems=stems,
link_in_signal=link_in_signal,
)
titles_by_id = {concept.concept_id: concept.title for concept in concepts}
matched = sum(1 for _, _, lexical in ranked if lexical > 0)