#!/bin/bash # route-selftest.sh - prove route.sh against the closed rubric row table, and # prove that what it emits is what board.sh parses back. Re-run after any edit # to route.sh or to the row table. ASCII only, bash 3.2 safe. # # The highest-value section here is 6, the round trip: route.sh WRITES the # next-cost value and board.sh READS it, and until now this repo owned only the # reader. A field with a reader and no writer drifts by construction - that is # the defect that put several competing spellings in circulation. One suite # now pins both ends, so a spelling change that breaks the parser fails here # instead of in the operator's eye three weeks later. set -u export LC_ALL=C DIR="$(cd "$(dirname "$0")" && pwd)" ROUTE="$DIR/route.sh" BOARD="$DIR/board.sh" PASS=0; FAIL=0 check() { if [ "$2" -eq 0 ]; then PASS=$((PASS+1)); echo " ok - $1"; else FAIL=$((FAIL+1)); echo " FAIL - $1"; fi; } # Multibyte building blocks (octal escapes keep this source ASCII). EMDASH="$(printf '\342\200\224')" HAND="$(printf '\360\237\221\211')" R="$ROUTE" # Shorthand: run route.sh with the four traits + a rationale, print one field. # $1..$4 traits, $5 field name. field() { "$R" --path "$1" --verification "$2" --reversibility "$3" --scope "$4" \ --rationale "selftest" 2>/dev/null | sed -n "s/^$5=//p" } echo "route-selftest" # --- 1. Every rubric row is reachable ------------------------------------- # A row that no trait combination can produce is dead policy. All six must fire. got="$(field known strong cheap local next-cost)" [ "$got" = "Sonnet 5/high" ]; check "row 1: known/strong/cheap/local -> Sonnet 5/high" $? got="$(field known weak cheap local next-cost)" [ "$got" = "Sonnet 5/xhigh" ]; check "row 2: weak verification -> Sonnet 5/xhigh" $? got="$(field partial strong cheap local next-cost)" [ "$got" = "Opus 5/high" ]; check "row 3: path=partial -> Opus 5/high" $? got="$(field known strong cheap cross-cutting next-cost)" [ "$got" = "Opus 5/xhigh" ]; check "row 4: scope=cross-cutting -> Opus 5/xhigh" $? got="$("$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --opus-xhigh-failed 2>/dev/null | sed -n 's/^next-cost=//p')" [ "$got" = "Fable 5/high" ]; check "row 5: opus-xhigh-failed -> Fable 5/high" $? got="$("$R" --path undetermined --verification strong --reversibility cheap --scope local \ --rationale x --opus-xhigh-failed 2>/dev/null | sed -n 's/^next-cost=//p')" [ "$got" = "Fable 5/xhigh" ]; check "row 6: failed + undetermined -> Fable 5/xhigh" $? # --- 2. Escalation is asymmetric ------------------------------------------ # One trait escalates; a downgrade needs ALL of them. Underkill costs one # session, overkill costs quota every session - so the cheap row must be hard # to reach and the expensive rows easy. got="$(field known strong one-way local next-cost)" [ "$got" = "Opus 5/xhigh" ]; check "one-way alone escalates to row 4" $? got="$(field known strong costly local next-cost)" [ "$got" = "Opus 5/high" ]; check "costly alone escalates to row 3" $? got="$(field known strong cheap multi-file next-cost)" [ "$got" = "Opus 5/high" ]; check "multi-file alone escalates to row 3" $? got="$(field known none cheap local next-cost)" [ "$got" = "Sonnet 5/xhigh" ]; check "verification=none alone escalates to row 2" $? got="$(field undetermined strong cheap local next-cost)" [ "$got" = "Opus 5/high" ]; check "path=undetermined stops at row 3, not row 4" $? # --- 3. The emitted vocabulary is CLOSED ---------------------------------- # Every one of the 3*3*3*3 trait combinations, in both history states, must # emit one of exactly six strings. This is what structurally prevents a # seventh spelling from ever entering circulation: not a convention, an # enumeration. The board line's drift was possible only because the field had # no writer with a closed range. VOCAB="|Sonnet 5/high|Sonnet 5/xhigh|Opus 5/high|Opus 5/xhigh|Fable 5/high|Fable 5/xhigh|" bad=0; n=0 for p in known partial undetermined; do for v in strong weak none; do for r in cheap costly one-way; do for s in local multi-file cross-cutting; do for h in "" "--opus-xhigh-failed"; do n=$((n+1)) out="$("$R" --path "$p" --verification "$v" --reversibility "$r" --scope "$s" \ --rationale x $h 2>/dev/null | sed -n 's/^next-cost=//p')" case "$VOCAB" in *"|$out|"*) ;; *) bad=$((bad+1)); echo " out-of-vocab: $p/$v/$r/$s [$out]" ;; esac done done done done done [ "$n" -eq 162 ] && [ "$bad" -eq 0 ] check "all 162 trait combinations emit one of the 6 rubric rows" $? # --- 4. Every trait is REQUIRED ------------------------------------------- # verification carries the most signal and is the one most often left out, so # a missing trait must be a hard error, never a silent default. A default here # would be indistinguishable from a scored value when the log is read back. "$R" --verification strong --reversibility cheap --scope local --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "missing --path exits 2" $? "$R" --path known --reversibility cheap --scope local --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "missing --verification exits 2" $? "$R" --path known --verification strong --scope local --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "missing --reversibility exits 2" $? "$R" --path known --verification strong --reversibility cheap --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "missing --scope exits 2" $? "$R" --path known --verification strong --reversibility cheap --scope local >/dev/null 2>&1 [ $? -eq 2 ]; check "missing --rationale exits 2" $? # An unscored trait must not be smuggled in as empty text either. "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale "" >/dev/null 2>&1 [ $? -eq 2 ]; check "empty --rationale exits 2" $? # --- 5. Invalid trait values are rejected --------------------------------- "$R" --path maybe --verification strong --reversibility cheap --scope local \ --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "invalid --path value exits 2" $? "$R" --path known --verification medium --reversibility cheap --scope local \ --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "invalid --verification value exits 2" $? "$R" --path known --verification strong --reversibility cheap --scope global \ --rationale x >/dev/null 2>&1 [ $? -eq 2 ]; check "invalid --scope value exits 2" $? "$R" --nonsense >/dev/null 2>&1 [ $? -eq 2 ]; check "unknown argument exits 2" $? # --- 6. ROUND TRIP: what route WRITES, board READS ------------------------ # The reason this repo owns the calculator at all. Splice route.sh's next-cost # into a board line, run the real board.sh over it, and require the KOST column # to show the same string back. Runs for all six rows. ROOT="$(mktemp -d)" MBOX="$(mktemp -d)" cleanup() { /bin/rm -rf "$ROOT" "$MBOX" 2>/dev/null; } trap cleanup EXIT rt_bad=0 rt_case() { # $1 repo name, $2 next-cost value mkdir -p "$ROOT/$1" && git -C "$ROOT/$1" init -q 2>/dev/null { echo "# STATE - $1" printf '## %s NESTE %s START HER\n' "$HAND" "$EMDASH" echo "" echo "" echo "**1.** en helt vanlig prosalinje her." } > "$ROOT/$1/STATE.md" } i=0 for combo in "known strong cheap local" "known weak cheap local" \ "partial strong cheap local" "known strong cheap cross-cutting"; do set -- $combo i=$((i+1)) cost="$(field "$1" "$2" "$3" "$4" next-cost)" rt_case "rt-$i" "$cost" done for h in 5 6; do if [ "$h" = "5" ]; then pp=known; else pp=undetermined; fi cost="$("$R" --path "$pp" --verification strong --reversibility cheap --scope local \ --rationale x --opus-xhigh-failed 2>/dev/null | sed -n 's/^next-cost=//p')" rt_case "rt-$h" "$cost" done OUT="$(CLAUDE_COORD_DIR="$MBOX" "$BOARD" --roots "$ROOT" 2>/dev/null)" for want in "Sonnet 5/high" "Sonnet 5/xhigh" "Opus 5/high" "Opus 5/xhigh" \ "Fable 5/high" "Fable 5/xhigh"; do printf '%s' "$OUT" | grep -q "$want" || { rt_bad=$((rt_bad+1)); echo " board lost: [$want]"; } done [ "$rt_bad" -eq 0 ]; check "round trip: board.sh parses back all 6 emitted values" $? # board.sh renders KOST with %-14s; a longer value shoves the whole row right # even though it parsed fine. Measure the widest string the table can emit - # not whatever the loop above happened to leave behind. widest=0 for v in "Sonnet 5/high" "Sonnet 5/xhigh" "Opus 5/high" "Opus 5/xhigh" \ "Fable 5/high" "Fable 5/xhigh"; do [ "${#v}" -gt "$widest" ] && widest="${#v}" done [ "$widest" -le 14 ]; check "widest emitted next-cost ($widest) fits the KOST column" $? # --- 7. The route line does not steal the NESTE column -------------------- # Measured before writing route.sh: board.sh's NESTE extractor skips blanks, # lines STARTING with '" echo "$LINE" echo "" echo "**1.** prosalinjen som skal overleve." } > "$ROOT/rt-neste/STATE.md" OUT2="$(CLAUDE_COORD_DIR="$MBOX" "$BOARD" --roots "$ROOT" 2>/dev/null)" printf '%s' "$OUT2" | grep -q 'prosalinjen som skal overleve' check "route line under NESTE leaves the prose in the NESTE column" $? printf '%s' "$LINE" | grep -q '^$' check "route line is a single-line HTML comment" $? # --- 8. Rationale is untrusted text on a comment line --------------------- # It is free text typed by a session and it lands inside an HTML comment on one # line. A newline splits the line (and hands the next line to the NESTE # extractor); a literal '-->' closes the comment early and dumps the rest into # the rendered STATE.md. Same line-oriented sanitizing the send side does. LINE2="$("$R" --path known --verification strong --reversibility cheap --scope local \ --rationale "$(printf 'first\nsecond')" 2>/dev/null | sed -n 's/^route-line=//p')" [ "$(printf '%s' "$LINE2" | wc -l | tr -d ' ')" -eq 0 ] check "newline in rationale does not split the route line" $? LINE3="$("$R" --path known --verification strong --reversibility cheap --scope local \ --rationale 'oops --> loose text' 2>/dev/null | sed -n 's/^route-line=//p')" [ "$(printf '%s' "$LINE3" | grep -c -- '-->')" -eq 1 ] check "'-->' in rationale cannot close the comment early" $? # --- 8b. The last-session record ------------------------------------------ # Rows 5 and 6 are history-dependent, so without a record of how the previous # session went they are dead policy. The record is also the cheap proxy for # whether the routing itself is any good: systematically high corrections on # row 1 means the cheap row is too easy to reach, systematically zero on row 4 # means escalation fires too readily. It gets a WRITER here for the same # reason next-cost needed one - a format with only a reader drifts. LAST="$("$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "Opus 5" --last-effort xhigh \ --last-completed no --last-corrections 3 2>/dev/null | sed -n 's/^route-last=//p')" printf '%s' "$LAST" | grep -q '^$' check "route-last line is emitted in the pinned single-line form" $? # Omitted record must emit no line at all rather than a half-filled one: a # blank record read back later is indistinguishable from a real measurement. out_norec="$("$R" --path known --verification strong --reversibility cheap \ --scope local --rationale x 2>/dev/null)" if printf '%s' "$out_norec" | grep -q '^route-last='; then rc=1; else rc=0; fi check "no route-last line when the record is omitted" "$rc" "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-completed maybe >/dev/null 2>&1 [ $? -eq 2 ]; check "invalid --last-completed exits 2" $? "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-corrections three >/dev/null 2>&1 [ $? -eq 2 ]; check "non-numeric --last-corrections exits 2" $? # The record is read back by the NEXT session to decide --opus-xhigh-failed, so # its model and effort are compared, not just displayed. Leaving them as free # text would rebuild the exact reader-versus-writer drift this script exists to # kill, one field over. Both are closed sets: the row table's three model names # and the verified effort levels. "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "opus 5" --last-effort xhigh \ --last-completed no --last-corrections 1 >/dev/null 2>&1 [ $? -eq 2 ]; check "--last-model rejects a non-rubric spelling" $? "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "Opus 5" --last-effort extreme \ --last-completed no --last-corrections 1 >/dev/null 2>&1 [ $? -eq 2 ]; check "--last-effort rejects a value outside the verified set" $? "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "Fable 5" --last-effort medium \ --last-completed yes --last-corrections 0 >/dev/null 2>&1 [ $? -eq 0 ]; check "--last-model/-effort accept every legal value" $? # The record is telemetry and must NOT silently fire the Fable rows. Only the # explicit flag does, because "an opus/xhigh session did not finish" also # covers context exhaustion, an operator interrupt and a block on another repo # - none of which mean the MODEL failed at this step. Fable runs without an # advisor, so this auto-escalation has to stay a judgement, not an inference. got="$("$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "Opus 5" --last-effort xhigh --last-completed no \ --last-corrections 4 2>/dev/null | sed -n 's/^next-cost=//p')" [ "$got" = "Sonnet 5/high" ]; check "a failed opus/xhigh record alone does NOT reach Fable" $? # A record is all four fields or none. A partial one emits `corrections=` with # nothing after it, which reads back later exactly like a measured zero. "$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x --last-model "Opus 5" --last-effort xhigh >/dev/null 2>&1 [ $? -eq 2 ]; check "a partial last-session record exits 2" $? # All three comment lines stacked under the heading must still leave the prose # in the NESTE column - that is the arrangement a real STATE.md ends up with. mkdir -p "$ROOT/rt-three" && git -C "$ROOT/rt-three" init -q 2>/dev/null { echo "# STATE - rt-three" printf '## %s NESTE %s START HER\n' "$HAND" "$EMDASH" echo "" echo "$LINE" echo "$LAST" echo "" # Kept well under board.sh's 38-char NESTE truncation: a longer line would # be cut mid-word and fail this grep for a reason that has nothing to do # with what is being tested. echo "**1.** tredje prosalinje." } > "$ROOT/rt-three/STATE.md" OUT3="$(CLAUDE_COORD_DIR="$MBOX" "$BOARD" --roots "$ROOT" 2>/dev/null)" printf '%s' "$OUT3" | grep -q 'tredje prosalinje' check "board + route + route-last stacked still yield prose in NESTE" $? # --- 9. Startup command and fallback -------------------------------------- # Two spellings of ONE decision: the rubric name for the board line, the CLI # alias for the command the operator pastes. They must never disagree. cmd="$(field partial strong cheap local command)" [ "$cmd" = "claude --model opus --effort high" ]; check "command mirrors the row (opus/high)" $? cmd="$(field known strong cheap local command)" [ "$cmd" = "claude --model sonnet --effort high" ]; check "command mirrors the row (sonnet/high)" $? # The rubric requires ALWAYS naming one row cheaper as the quota fallback. fb="$(field partial strong cheap local fallback)" [ "$fb" = "Sonnet 5/xhigh" ]; check "fallback is exactly one row cheaper" $? fb="$(field known strong cheap local fallback)" [ "$fb" = "Sonnet 5/high" ]; check "row 1 fallback floors at row 1, never below" $? # The fallback needs its own pasteable command or the operator translates by # hand at exactly the moment they are under quota pressure. fbc="$(field partial strong cheap local fallback-command)" [ "$fbc" = "claude --model sonnet --effort xhigh" ]; check "fallback ships its own command" $? # --- 10. The command carries no 'cd' -------------------------------------- # One repo per terminal tab: a startup command prefixed with cd is wrong by # construction, and another directory means another tab. out="$("$R" --path known --verification strong --reversibility cheap --scope local \ --rationale x 2>/dev/null)" if printf '%s' "$out" | grep -q 'cd '; then rc=1; else rc=0; fi check "no emitted command contains a cd prefix" "$rc" # --- 11. Effort and model values are the VERIFIED sets -------------------- # Effort levels are pinned in this marketplace at # config-audit/scanners/settings-validator.mjs (low|medium|high|xhigh|max). # Model aliases are whatever the INSTALLED claude accepts - never hardcoded # without a gate, because an alias that stops resolving turns every emitted # command into a paste that fails. efforts="$(printf '%s' "$out" | sed -n 's/^command=claude --model [a-z]* --effort //p')" case "|low|medium|high|xhigh|max|" in *"|$efforts|"*) rc=0 ;; *) rc=1 ;; esac check "emitted effort is in the verified effort set" "$rc" if command -v claude >/dev/null 2>&1; then # Match ONLY the quoted alias as --model documents it. An unanchored grep for # the bare word would hit "opus" anywhere in the help text and pass even if # --model stopped accepting the alias entirely - a gate that reports success # without testing anything, which is worse than no gate. HELPTXT="$(claude --help 2>&1)" miss="" for alias in fable opus sonnet; do printf '%s' "$HELPTXT" | grep -q "'$alias'" || miss="$miss $alias" done [ -z "$miss" ]; check "installed claude documents the aliases route emits:${miss:- all three}" $? else echo " skip - claude not on PATH, model alias gate not run" fi # --- 12. One spec, and it is the row table -------------------------------- # The last board fix closed a defect whose root cause was this repo's own # --help being the SECOND spec for next-cost. route.sh must not reopen it: its # help may describe the row table (it owns it now) but must not restate the # board line grammar, which board.sh --help owns. HELP="$("$R" --help 2>/dev/null)"; rc=$? [ "$rc" -eq 0 ]; check "--help exits 0" $? printf '%s' "$HELP" | grep -q 'Sonnet 5/high' check "--help shows the canonical rubric spelling" $? if printf '%s' "$HELP" | grep -qE '(sonnet|opus|fable) ?5?/(high|xhigh)'; then rc=1; else rc=0; fi check "--help shows no versionless model example" "$rc" if printf '%s' "$HELP" | grep -q 'blocked-on='; then rc=1; else rc=0; fi check "--help does not restate the board line grammar" "$rc" echo "" echo "route-selftest: $PASS passed, $FAIL failed" [ "$FAIL" -eq 0 ] || exit 1 exit 0