#!/bin/bash # route-selftest.sh - prove route.sh against the closed rubric row table, and # prove that what it emits is what board.sh parses back. Re-run after any edit # to route.sh or to the row table. ASCII only, bash 3.2 safe. # # The highest-value section here is 6, the round trip: route.sh WRITES the # next-cost value and board.sh READS it, and until now this repo owned only the # reader. A field with a reader and no writer drifts by construction - that is # the defect that put several competing spellings in circulation. One suite # now pins both ends, so a spelling change that breaks the parser fails here # instead of in the operator's eye three weeks later. set -u export LC_ALL=C DIR="$(cd "$(dirname "$0")" && pwd)" ROUTE="$DIR/route.sh" BOARD="$DIR/board.sh" PASS=0; FAIL=0 check() { if [ "$2" -eq 0 ]; then PASS=$((PASS+1)); echo " ok - $1"; else FAIL=$((FAIL+1)); echo " FAIL - $1"; fi; } # Multibyte building blocks (octal escapes keep this source ASCII). EMDASH="$(printf '\342\200\224')" HAND="$(printf '\360\237\221\211')" R="$ROUTE" # Shorthand: run route.sh with the four traits + a rationale, print one field. # $1..$4 traits, $5 field name. field() { "$R" --path "$1" --verification "$2" --reversibility "$3" --scope "$4" \ --rationale "selftest" 2>/dev/null | sed -n "s/^$5=//p" } echo "route-selftest" # --- 1. Every calculator row is reachable ----------------------------------- # A row that no trait combination can produce is dead policy. All four rows # this calculator can output must fire from traits alone. got="$(field known strong cheap local next-cost)" [ "$got" = "Sonnet 5/high" ]; check "row 1: known/strong/cheap/local -> Sonnet 5/high" $? got="$(field known weak cheap local next-cost)" [ "$got" = "Sonnet 5/xhigh" ]; check "row 2: weak verification -> Sonnet 5/xhigh" $? got="$(field partial strong cheap local next-cost)" [ "$got" = "Opus 5/high" ]; check "row 3: path=partial -> Opus 5/high" $? got="$(field known strong cheap cross-cutting next-cost)" [ "$got" = "Opus 5/xhigh" ]; check "row 4: scope=cross-cutting -> Opus 5/xhigh" $? # Rows 5-6 (Fable) are the operator's hand-written override (policy decision # 2026-08-06), never a rubric outcome - the calculator's output range is # closed at row 4. The flag that used to gate them into reach is gone # outright, not merely disarmed: a caller passing it gets the same "unknown # argument" as any other typo. "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --opus-xhigh-failed >/dev/null 2>&1 [ $? -eq 2 ]; check "--opus-xhigh-failed is gone: unknown argument, not a route to Fable" $? # --- 2. Escalation is asymmetric ------------------------------------------ # One trait escalates; a downgrade needs ALL of them. Underkill costs one # session, overkill costs quota every session - so the cheap row must be hard # to reach and the expensive rows easy. got="$(field known strong one-way local next-cost)" [ "$got" = "Opus 5/xhigh" ]; check "one-way alone escalates to row 4" $? got="$(field known strong costly local next-cost)" [ "$got" = "Opus 5/high" ]; check "costly alone escalates to row 3" $? got="$(field known strong cheap multi-file next-cost)" [ "$got" = "Opus 5/high" ]; check "multi-file alone escalates to row 3" $? got="$(field known none cheap local next-cost)" [ "$got" = "Sonnet 5/xhigh" ]; check "verification=none alone escalates to row 2" $? got="$(field undetermined strong cheap local next-cost)" [ "$got" = "Opus 5/high" ]; check "path=undetermined stops at row 3, not row 4" $? # --- 3. The emitted vocabulary is CLOSED ---------------------------------- # Every one of the 3*3*3*3 trait combinations must emit one of exactly FOUR # strings - the calculator's whole output range now that Fable is reached # only by a hand-written operator override, never by this script. This is # what structurally prevents a fifth spelling from ever entering circulation: # not a convention, an enumeration. The board line's drift was possible only # because the field had no writer with a closed range. VOCAB="|Sonnet 5/high|Sonnet 5/xhigh|Opus 5/high|Opus 5/xhigh|" bad=0; n=0 for p in known partial undetermined; do for v in strong weak none; do for r in cheap costly one-way; do for s in local multi-file cross-cutting; do n=$((n+1)) out="$("$R" --path "$p" --verification "$v" --reversibility "$r" --scope "$s" \ --rationale x 2>/dev/null | sed -n 's/^next-cost=//p')" case "$VOCAB" in *"|$out|"*) ;; *) bad=$((bad+1)); echo " out-of-vocab: $p/$v/$r/$s [$out]" ;; esac done done done done [ "$n" -eq 81 ] && [ "$bad" -eq 0 ] check "all 81 trait combinations emit one of the 4 calculator rows" $? # --- 4. Every trait is REQUIRED ------------------------------------------- # verification carries the most signal and is the one most often left out, so # a missing trait must be a hard error, never a silent default. A default here # would be indistinguishable from a scored value when the log is read back. "$R" --verification strong --reversibility cheap --scope local --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "missing --path exits 2" $? "$R" --path known --reversibility cheap --scope local --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "missing --verification exits 2" $? "$R" --path known --verification strong --scope local --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "missing --reversibility exits 2" $? "$R" --path known --verification strong --reversibility cheap --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "missing --scope exits 2" $? "$R" --path known --verification strong --reversibility cheap --scope local >/dev/null 2>&1 [ $? -eq 2 ]; check "missing --rationale exits 2" $? # An unscored trait must not be smuggled in as empty text either. "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale "" >/dev/null 2>&1 [ $? -eq 2 ]; check "empty --rationale exits 2" $? # --- 5. Invalid trait values are rejected --------------------------------- "$R" --path maybe --verification strong --reversibility cheap --scope local \ --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "invalid --path value exits 2" $? "$R" --path known --verification medium --reversibility cheap --scope local \ --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "invalid --verification value exits 2" $? "$R" --path known --verification strong --reversibility cheap --scope global \ --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "invalid --scope value exits 2" $? "$R" --nonsense >/dev/null 2>&1 [ $? -eq 2 ]; check "unknown argument exits 2" $? # --- 6. ROUND TRIP: what route WRITES, board READS ------------------------ # The reason this repo owns the calculator at all. Splice route.sh's next-cost # into a board line, run the real board.sh over it, and require the KOST column # to show the same string back. Runs for all six rows. ROOT="$(mktemp -d)" MBOX="$(mktemp -d)" RL_ROOT="" cleanup() { /bin/rm -rf "$ROOT" "$MBOX" ${RL_ROOT:+"$RL_ROOT"} 2>/dev/null; } trap cleanup EXIT rt_bad=0 rt_case() { # $1 repo name, $2 next-cost value mkdir -p "$ROOT/$1" && git -C "$ROOT/$1" init -q 2>/dev/null { echo "# STATE - $1" printf '## %s NESTE %s START HER\n' "$HAND" "$EMDASH" echo "" echo "" echo "**1.** en helt vanlig prosalinje her." } > "$ROOT/$1/STATE.md" } i=0 for combo in "known strong cheap local" "known weak cheap local" \ "partial strong cheap local" "known strong cheap cross-cutting"; do set -- $combo i=$((i+1)) cost="$(field "$1" "$2" "$3" "$4" next-cost)" rt_case "rt-$i" "$cost" done # Rows 5-6 are never emitted by route.sh any more (policy decision # 2026-08-06), but board.sh must still parse them back when the operator # hand-writes a Fable board line - that is exactly the path that replaces the # removed rubric outcome, so the literal strings are spliced in directly here # rather than produced by "$R". rt_case "rt-5" "Fable 5/high" rt_case "rt-6" "Fable 5/xhigh" # A hand-written Fable 5.1 board line. next-cost extraction is free text, so # board.sh parses the point release back unchanged - pinned here so a later # narrowing of that extraction fails in this suite rather than in the # operator's eye. MEASURED GAP, stated rather than closed: at 15 characters it # overflows the %-14s KOST column and shifts the rest of that row one column # right. That is a board.sh rendering change nobody ordered in this session, so # it is reported to .claude, not fixed here - which is also why "Fable # 5.1/xhigh" is deliberately absent from the widest-value loop below. Adding it # there would go red, and the red would be the unfixed gap, not a broken test. rt_case "rt-51" "Fable 5.1/xhigh" OUT="$(CLAUDE_COORD_DIR="$MBOX" "$BOARD" --roots "$ROOT" 2>/dev/null)" for want in "Sonnet 5/high" "Sonnet 5/xhigh" "Opus 5/high" "Opus 5/xhigh" \ "Fable 5/high" "Fable 5/xhigh"; do printf '%s' "$OUT" | grep -q "$want" || { rt_bad=$((rt_bad+1)); echo " board lost: [$want]"; } done [ "$rt_bad" -eq 0 ]; check "round trip: board.sh parses back all 6 emitted values" $? printf '%s' "$OUT" | grep -q 'Fable 5\.1/xhigh' check "board parses back a hand-written Fable 5.1 next-cost" $? # board.sh renders KOST with %-14s; a longer value shoves the whole row right # even though it parsed fine. Measure the widest string the table can emit - # not whatever the loop above happened to leave behind. widest=0 for v in "Sonnet 5/high" "Sonnet 5/xhigh" "Opus 5/high" "Opus 5/xhigh" \ "Fable 5/high" "Fable 5/xhigh"; do [ "${#v}" -gt "$widest" ] && widest="${#v}" done [ "$widest" -le 14 ]; check "widest emitted next-cost ($widest) fits the KOST column" $? # --- 7. The route line does not steal the NESTE column -------------------- # Measured before writing route.sh: board.sh's NESTE extractor skips blanks, # lines STARTING with '" echo "$LINE" echo "" echo "**1.** prosalinjen som skal overleve." } > "$ROOT/rt-neste/STATE.md" OUT2="$(CLAUDE_COORD_DIR="$MBOX" "$BOARD" --roots "$ROOT" 2>/dev/null)" printf '%s' "$OUT2" | grep -q 'prosalinjen som skal overleve' check "route line under NESTE leaves the prose in the NESTE column" $? printf '%s' "$LINE" | grep -q '^$' check "route line is a single-line HTML comment" $? # --- 8. Rationale is untrusted text on a comment line --------------------- # It is free text typed by a session and it lands inside an HTML comment on one # line. A newline splits the line (and hands the next line to the NESTE # extractor); a literal '-->' closes the comment early and dumps the rest into # the rendered STATE.md. Same line-oriented sanitizing the send side does. LINE2="$("$R" --path known --verification strong --reversibility cheap --scope local \ --rationale "$(printf 'first\nsecond')" 2>/dev/null | sed -n 's/^route-line=//p')" [ "$(printf '%s' "$LINE2" | wc -l | tr -d ' ')" -eq 0 ] check "newline in rationale does not split the route line" $? LINE3="$("$R" --path known --verification strong --reversibility cheap --scope local \ --rationale 'oops --> loose text' 2>/dev/null | sed -n 's/^route-line=//p')" [ "$(printf '%s' "$LINE3" | grep -c -- '-->')" -eq 1 ] check "'-->' in rationale cannot close the comment early" $? # --- 8b. The last-session record ------------------------------------------ # The record is the cheap proxy for whether the routing itself is any good: # systematically high corrections on row 1 means the cheap row is too easy to # reach, systematically zero on row 4 means escalation fires too readily. It # gets a WRITER here for the same reason next-cost needed one - a format with # only a reader drifts. It is pure telemetry - it never changes what the # calculator outputs, Fable rows included, which are unreachable through it # entirely now. LAST="$("$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "Opus 5" --last-effort xhigh \ --last-completed no --last-corrections 3 2>/dev/null | sed -n 's/^route-last=//p')" printf '%s' "$LAST" | grep -q '^$' check "route-last line is emitted in the pinned single-line form" $? # Omitted record must emit no line at all rather than a half-filled one: a # blank record read back later is indistinguishable from a real measurement. out_norec="$("$R" --path known --verification strong --reversibility cheap \ --scope local --rationale x 2>/dev/null)" if printf '%s' "$out_norec" | grep -q '^route-last='; then rc=1; else rc=0; fi check "no route-last line when the record is omitted" "$rc" "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-completed maybe >/dev/null 2>&1 [ $? -eq 2 ]; check "invalid --last-completed exits 2" $? "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-corrections three >/dev/null 2>&1 [ $? -eq 2 ]; check "non-numeric --last-corrections exits 2" $? # The record is read back by the NEXT session as evidence months from now, so # its model and effort are compared, not just displayed. Leaving them as free # text would rebuild the exact reader-versus-writer drift this script exists to # kill, one field over. Both are closed sets: the row table's three model names # and the verified effort levels. "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "opus 5" --last-effort xhigh \ --last-completed no --last-corrections 1 >/dev/null 2>&1 [ $? -eq 2 ]; check "--last-model rejects a non-rubric spelling" $? "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "Opus 5" --last-effort extreme \ --last-completed no --last-corrections 1 >/dev/null 2>&1 [ $? -eq 2 ]; check "--last-effort rejects a value outside the verified set" $? "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "Fable 5" --last-effort medium \ --last-completed yes --last-corrections 0 >/dev/null 2>&1 [ $? -eq 0 ]; check "--last-model/-effort accept every legal value" $? # Fable 5.1 shipped 2026-09-01 and the closed set refused it, so a session that # actually ran it could not record what it ran: the record was either omitted # or LIED, and a lied record reads back months later as a measurement. The set # is WIDENED, never replaced by form validation - the check below is what makes # that choice machine-verified instead of prose. "Fable 5" stays legal for a # reason stronger than the one STATE.md on this machine that still carries it: # route.sh's OWN row table spells rows 5-6 "Fable 5/high" and "Fable 5/xhigh", # so dropping it would make the script refuse to record a value its own spec # names. The check above this one is what goes red if anyone drops it. "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "Fable 5.1" --last-effort xhigh \ --last-completed yes --last-corrections 0 >/dev/null 2>&1 [ $? -eq 0 ]; check "--last-model accepts the Fable 5.1 point release" $? # The set is still CLOSED after being widened, and this is the whole cost of # NOT switching to form validation. A pattern like " [.]" # would accept every line below, and would stop catching a version that does # not exist - which reads back later as evidence that a model ran when it never # shipped. That is the positive-looking null this repo refuses everywhere else. fable_bad=0 for bad in "Fable 5.2" "Fable 6" "Fable 5.10"; do "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "$bad" --last-effort xhigh \ --last-completed yes --last-corrections 0 >/dev/null 2>&1 [ $? -eq 2 ] || { fable_bad=$((fable_bad+1)); echo " accepted a model that does not exist: [$bad]"; } done [ "$fable_bad" -eq 0 ]; check "the model set stays CLOSED after Fable 5.1 (no form validation)" $? # Accepting the value is not the same as RECORDING it. The record is what the # next session reads back, so the emitted line must carry the point release # verbatim rather than collapsing it to the family name. LAST51="$("$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "Fable 5.1" --last-effort xhigh \ --last-completed yes --last-corrections 0 2>/dev/null | sed -n 's/^route-last=//p')" printf '%s' "$LAST51" | grep -q '^$' check "route-last carries Fable 5.1 verbatim into the emitted line" $? # The record is telemetry and must NOT silently change what the calculator # outputs - a "completed=no" record describes what happened, and covers # context exhaustion, an operator interrupt and a block on another repo just # as much as an actual model failure. Reading it as an inference would revive # exactly the escalation path the removed --opus-xhigh-failed flag used to # gate deliberately. got="$("$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "Opus 5" --last-effort xhigh --last-completed no \ --last-corrections 4 2>/dev/null | sed -n 's/^next-cost=//p')" [ "$got" = "Sonnet 5/high" ]; check "a failed-session record alone does not change the routing outcome" $? # A record is all four fields or none. A partial one emits `corrections=` with # nothing after it, which reads back later exactly like a measured zero. "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "Opus 5" --last-effort xhigh >/dev/null 2>&1 [ $? -eq 2 ]; check "a partial last-session record exits 2" $? # All three comment lines stacked under the heading must still leave the prose # in the NESTE column - that is the arrangement a real STATE.md ends up with. mkdir -p "$ROOT/rt-three" && git -C "$ROOT/rt-three" init -q 2>/dev/null { echo "# STATE - rt-three" printf '## %s NESTE %s START HER\n' "$HAND" "$EMDASH" echo "" echo "$LINE" echo "$LAST" echo "" # Kept well under board.sh's 38-char NESTE truncation: a longer line would # be cut mid-word and fail this grep for a reason that has nothing to do # with what is being tested. echo "**1.** tredje prosalinje." } > "$ROOT/rt-three/STATE.md" OUT3="$(CLAUDE_COORD_DIR="$MBOX" "$BOARD" --roots "$ROOT" 2>/dev/null)" printf '%s' "$OUT3" | grep -q 'tredje prosalinje' check "board + route + route-last stacked still yield prose in NESTE" $? # --- 9. Startup command and fallback -------------------------------------- # Two spellings of ONE decision: the rubric name for the board line, the CLI # alias for the command the operator pastes. They must never disagree. # These assert the WHOLE string, so no flag can be added back to the emitted # command without a deliberate edit here - the exact-match is the tripwire that # keeps section 14's absence rule from being widened around. cmd="$(field partial strong cheap local command)" [ "$cmd" = "claude --model opus --effort high" ]; check "command mirrors the row (opus/high)" $? cmd="$(field known strong cheap local command)" [ "$cmd" = "claude --model sonnet --effort high" ]; check "command mirrors the row (sonnet/high)" $? # The rubric requires ALWAYS naming one row cheaper as the quota fallback. fb="$(field partial strong cheap local fallback)" [ "$fb" = "Sonnet 5/xhigh" ]; check "fallback is exactly one row cheaper" $? fb="$(field known strong cheap local fallback)" [ "$fb" = "Sonnet 5/high" ]; check "row 1 fallback floors at row 1, never below" $? # The fallback needs its own pasteable command or the operator translates by # hand at exactly the moment they are under quota pressure. fbc="$(field partial strong cheap local fallback-command)" [ "$fbc" = "claude --model sonnet --effort xhigh" ]; check "fallback ships its own command" $? # --- 10. The command carries no 'cd' -------------------------------------- # One repo per terminal tab: a startup command prefixed with cd is wrong by # construction, and another directory means another tab. out="$("$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x 2>/dev/null)" if printf '%s' "$out" | grep -q 'cd '; then rc=1; else rc=0; fi check "no emitted command contains a cd prefix" "$rc" # --- 11. Effort and model values are the VERIFIED sets -------------------- # Effort levels are pinned in this marketplace at # config-audit/scanners/settings-validator.mjs (low|medium|high|xhigh|max). # Model aliases are whatever the INSTALLED claude accepts - never hardcoded # without a gate, because an alias that stops resolving turns every emitted # command into a paste that fails. # Capture the effort TOKEN only. The command ends at the effort today, but the # pattern stays tail-tolerant: a match that swallowed whatever a later flag # added would report a valid effort as invalid. efforts="$(printf '%s' "$out" | sed -n 's/^command=claude --model [a-z]* --effort \([a-z]*\).*/\1/p')" case "|low|medium|high|xhigh|max|" in *"|$efforts|"*) rc=0 ;; *) rc=1 ;; esac check "emitted effort is in the verified effort set" "$rc" if command -v claude >/dev/null 2>&1; then # Match ONLY the quoted alias as --model documents it. An unanchored grep for # the bare word would hit "opus" anywhere in the help text and pass even if # --model stopped accepting the alias entirely - a gate that reports success # without testing anything, which is worse than no gate. HELPTXT="$(claude --help 2>&1)" miss="" for alias in fable opus sonnet; do printf '%s' "$HELPTXT" | grep -q "'$alias'" || miss="$miss $alias" done [ -z "$miss" ]; check "installed claude documents the aliases route emits:${miss:- all three}" $? else echo " skip - claude not on PATH, model alias gate not run" fi # --- 12. One spec, and it is the row table -------------------------------- # The last board fix closed a defect whose root cause was this repo's own # --help being the SECOND spec for next-cost. route.sh must not reopen it: its # help may describe the row table (it owns it now) but must not restate the # board line grammar, which board.sh --help owns. HELP="$("$R" --help 2>/dev/null)"; rc=$? [ "$rc" -eq 0 ]; check "--help exits 0" $? printf '%s' "$HELP" | grep -q 'Sonnet 5/high' check "--help shows the canonical rubric spelling" $? if printf '%s' "$HELP" | grep -qE '(sonnet|opus|fable) ?5?/(high|xhigh)'; then rc=1; else rc=0; fi check "--help shows no versionless model example" "$rc" if printf '%s' "$HELP" | grep -q 'blocked-on='; then rc=1; else rc=0; fi check "--help does not restate the board line grammar" "$rc" # --- 13. Where --last-effort comes from ----------------------------------- # The record exists to make the policy falsifiable, which it only is if the # effort field is MEASURED. Two earlier sources both measured the wrong thing: # the previous board line holds what was PRESCRIBED, and asking the operator # launders that same prescription through a human who is reading it off the # startup command they typed. $CLAUDE_EFFORT is what the session actually # resolved - it is documented as the CURRENT effort level and is exported into # every tool-use context, which is why a Bash call can read it at all. # # The trap this section exists to pin: skill frontmatter can set `effort`, and # frontmatter overrides the session level while that skill is active. An # `effort:` field in route's own SKILL.md would therefore make the reading # report the SKILL's effort and not the session's - a measurement quietly # measuring itself, with nothing in the output to show it happened. SKILL="$DIR/../skills/route/SKILL.md" [ -f "$SKILL" ]; check "route SKILL.md is where the selftest expects it" $? # Frontmatter only: the body must be free to DISCUSS effort at length. FM="$(awk 'NR==1 && /^---$/ {f=1; next} f && /^---$/ {exit} f {print}' "$SKILL" 2>/dev/null)" if printf '%s' "$FM" | grep -q '^effort:'; then rc=1; else rc=0; fi check "route SKILL.md declares no effort: frontmatter field" "$rc" grep -q 'CLAUDE_EFFORT' "$SKILL" check "route SKILL.md names CLAUDE_EFFORT as the --last-effort source" $? grep -q 'board line' "$SKILL" check "route SKILL.md still warns off the previous board line" $? # route.sh carried the claim as a documented premise. It was true when written # and is not any more, so it must not survive as a comment that reads like a # measured fact three weeks from now. FLAT="$(tr '\n' ' ' < "$R" | sed 's/#//g' | tr -s ' ')" if printf '%s' "$FLAT" | grep -q 'is not observable from inside'; then rc=1; else rc=0; fi check "route.sh no longer claims effort is unobservable from inside" "$rc" grep -q 'CLAUDE_EFFORT' "$R" check "route.sh names the source the caller should measure from" $? # --- 14. The advisor is NOT the rubric's to emit -------------------------- # Struck by operator decision 2026-09-12 # (~/.claude/docs/2026-09-12-helhetlig-vurdering-arbeidssystemet.md, cut row 3). # The rule it replaces fired on two needs - Sonnet rows always, Opus rows at # costly|one-way stakes - and read well. What killed it was a measurement, not # a change of taste: of 54 dispatches the PM issued 08.-12.09, ZERO carried # --advisor opus, because the operator starts sessions by hand and pastes the # model and effort, not the whole line. A rule nothing honours is not a policy, # it is a claim about the world that the world disagrees with - and this repo's # own standing rule is that an emitted value must be evidence, never decoration. # # So the calculator emits no advisor at all, and the advisor becomes what it # already was in practice: an operator decision per session. That is a strictly # SAFER direction for the one thing the old rule protected - the quota fallback # is still one row cheaper, it just no longer implies a capability lift nobody # was taking. # # Pinned as an ABSENCE over the whole trait space rather than on four sampled # rows, because the claim is that no path emits it - the same "no write path # exists" argument the channel split uses. An absence check is worthless # without a known-positive control, so the sweep's own grep is proven able to # find a planted advisor before it is trusted to report none. adv="$(field known strong cheap local command)" if printf '%s' "$adv" | grep -q -- '--advisor'; then rc=1; else rc=0; fi check "row 1 (Sonnet/high) carries no advisor" "$rc" adv="$(field known weak cheap local command)" if printf '%s' "$adv" | grep -q -- '--advisor'; then rc=1; else rc=0; fi check "row 2 (Sonnet/xhigh) carries no advisor" "$rc" adv="$(field known strong costly local command)" if printf '%s' "$adv" | grep -q -- '--advisor'; then rc=1; else rc=0; fi check "reversibility=costly no longer pulls an advisor onto an Opus row" "$rc" adv="$(field known strong one-way local command)" if printf '%s' "$adv" | grep -q -- '--advisor'; then rc=1; else rc=0; fi check "reversibility=one-way no longer pulls an advisor onto an Opus row" "$rc" # The fallback is the half the old rule called load-bearing, so it is pinned # in its own right: dropping a row under quota pressure must not reintroduce # the flag by the back door. adv="$(field known strong one-way local fallback-command)" if printf '%s' "$adv" | grep -q -- '--advisor'; then rc=1; else rc=0; fi check "the row-4 fallback carries no advisor either" "$rc" adv="$(field partial strong cheap local fallback-command)" if printf '%s' "$adv" | grep -q -- '--advisor'; then rc=1; else rc=0; fi check "an Opus row falling back to a Sonnet row carries no advisor" "$rc" # THE SWEEP: every trait combination the calculator accepts, every line of # output. 81 combinations, so a rule surviving on one unsampled corner cannot # hide. Accumulated into one string and grepped once - a per-combination check # would add 81 lines to the summary and drown the rest of the suite. sweep="" for s_path in known partial undetermined; do for s_ver in strong weak none; do for s_rev in cheap costly one-way; do for s_sc in local multi-file cross-cutting; do sweep="$sweep $("$R" --path "$s_path" --verification "$s_ver" --reversibility "$s_rev" \ --scope "$s_sc" --rationale x 2>/dev/null)" done done done done if printf '%s' "$sweep" | grep -q -- '--advisor'; then rc=1; else rc=0; fi check "no advisor in any output over all 81 trait combinations" "$rc" # Known-positive control: the grep above reports an absence, so it must be # shown able to report a presence. Without this the sweep would pass just as # happily against an empty string. if printf '%s' "$sweep claude --advisor opus" | grep -q -- '--advisor'; then rc=0; else rc=1; fi check "control: the sweep's grep does find a planted advisor" "$rc" # The help text is the rubric's published form, so the rule has to leave there # too - a doc line nothing tests is a doc line that drifts, and a struck rule # still written down is worse than one never removed: it reads as current. # The literal flag string is absent from route.sh ENTIRELY, including the # paragraph that records what was struck - that paragraph names "an opus # advisor flag" in words on purpose. A blunt grep cannot tell a description # from a specification (the same reason the cache assertion in coord-selftest # runs on an extracted path rather than the whole file), and keeping the # string out is cheaper than teaching the check to read prose. Do not # "restore the quote" here. HELPOUT="$("$R" --help 2>/dev/null)" if printf '%s' "$HELPOUT" | grep -q -- '--advisor opus'; then rc=1; else rc=0; fi check "--help no longer documents emitting --advisor opus" "$rc" if printf '%s' "$HELPOUT" | grep -q 'THE ADVISOR is emitted'; then rc=1; else rc=0; fi check "--help no longer carries the advisor rule block" "$rc" # Removing the rule silently would leave a reader guessing whether the advisor # is forbidden, forgotten, or someone else's. It is the third, and the help # says which. printf '%s' "$HELPOUT" | grep -q 'advisor is an operator decision per session' check "--help states whose decision the advisor is instead" $? grep -q 'advisor is an operator decision per session' "$R" check "route.sh itself carries that sentence, not just its help output" $? # --- 14b. Old route lines still parse ------------------------------------ # Backward compatibility, pinned rather than assumed. Measured on the real # tree the day the rule was struck: 0 of 48 `\n' printf '\n' printf 'legacy next step\n' } > "$RL_ROOT/repo-legacy/STATE.md" RL_OUT="$("$BOARD" --roots "$RL_ROOT" --plan 2>/dev/null)" printf '%s' "$RL_OUT" | grep -q '^command=claude --model sonnet --effort high$' check "a route line carrying a legacy advisor= field still yields a command" $? if printf '%s' "$RL_OUT" | grep -q -- '--advisor'; then rc=1; else rc=0; fi check "and the command derived from it carries no advisor" "$rc" echo "" echo "route-selftest: $PASS passed, $FAIL failed" [ "$FAIL" -eq 0 ] || exit 1 exit 0