feat(s31): semretrieval thread-pinned Embedder seam + fake embedder + cosine
This commit is contained in:
parent
8a8c4f0fca
commit
466ff969fd
2 changed files with 259 additions and 0 deletions
133
tests/test_semretrieval.py
Normal file
133
tests/test_semretrieval.py
Normal file
|
|
@ -0,0 +1,133 @@
|
|||
"""S3.1 — semantic retrieval unit tests (SC4 here; SC1/SC3/SC7 added by later steps).
|
||||
|
||||
The embedder must be a PURE function of the structured features: identical features yield a
|
||||
byte-identical float64 vector in this process AND in a fresh process started with a different
|
||||
``PYTHONHASHSEED`` — the cross-process leg is what rules out Python's salted builtin ``hash()``
|
||||
(the exact failure mode that would make retrieval order drift between runs).
|
||||
|
||||
Components are non-negative, so ``cosine`` stays in ``[0, 1]`` and blends with the structural
|
||||
score (also ``[0, 1]``) without a scale mismatch.
|
||||
|
||||
Pattern: tests/test_verdicts.py (determinism) + tests/test_retrieval.py (fixtures).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import subprocess
|
||||
import sys
|
||||
import types
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
from portfolio_optimiser.semretrieval import EMBED_DIM, FakeEmbedder, cosine
|
||||
|
||||
_SRC = Path(__file__).resolve().parents[1] / "src"
|
||||
|
||||
|
||||
def _features(
|
||||
codes: frozenset[str] = frozenset({"05.2", "03.1"}),
|
||||
measure_type: str = "scope_reduction",
|
||||
saving: float = 220_000.0,
|
||||
description: str = "asphalt base course reduction near school",
|
||||
) -> types.SimpleNamespace:
|
||||
"""Duck-typed stand-in for ``ProposalFeatures`` — semretrieval never imports ``verdicts``
|
||||
at runtime, so the unit tests do not need the real (MAF-bound) type either."""
|
||||
return types.SimpleNamespace(
|
||||
affected_codes=codes,
|
||||
measure_type=measure_type,
|
||||
claimed_saving_nok=saving,
|
||||
description=description,
|
||||
)
|
||||
|
||||
|
||||
def test_fake_embedder_is_pure() -> None:
|
||||
embedder = FakeEmbedder()
|
||||
assert np.array_equal(embedder(_features()), embedder(_features()))
|
||||
|
||||
|
||||
def test_fake_embedder_has_fixed_float64_dimension() -> None:
|
||||
vec = FakeEmbedder()(_features())
|
||||
assert vec.shape == (EMBED_DIM,)
|
||||
assert vec.dtype == np.dtype("<f8")
|
||||
|
||||
|
||||
def test_fake_embedder_is_insensitive_to_code_set_ordering() -> None:
|
||||
"""``affected_codes`` is a set — the canonical string sorts it, so insertion order
|
||||
cannot leak into the vector."""
|
||||
a = FakeEmbedder()(_features(codes=frozenset({"05.2", "03.1"})))
|
||||
b = FakeEmbedder()(_features(codes=frozenset({"03.1", "05.2"})))
|
||||
assert np.array_equal(a, b)
|
||||
|
||||
|
||||
def test_distinct_features_give_distinct_vectors() -> None:
|
||||
embedder = FakeEmbedder()
|
||||
baseline = embedder(_features())
|
||||
for variant in (
|
||||
_features(codes=frozenset({"09.1"})),
|
||||
_features(measure_type="rate_renegotiation"),
|
||||
_features(saving=900_000.0), # different magnitude bucket
|
||||
_features(description="totally different wording"),
|
||||
):
|
||||
assert not np.array_equal(baseline, embedder(variant))
|
||||
|
||||
|
||||
def test_fake_embedder_components_are_non_negative() -> None:
|
||||
"""Non-negative components keep ``cosine`` in [0, 1], matching the structural score's
|
||||
range — no scale mismatch when the two are blended by HybridRanker."""
|
||||
vec = FakeEmbedder()(_features())
|
||||
assert bool((vec >= 0.0).all())
|
||||
|
||||
|
||||
def test_fake_embedder_is_bit_stable_across_processes(tmp_path: Path) -> None:
|
||||
"""RED for any ``hash()``-based implementation: the child runs with a different
|
||||
``PYTHONHASHSEED``, so a salted hash would produce different bytes."""
|
||||
embedder = FakeEmbedder()
|
||||
here = tmp_path / "here.npy"
|
||||
np.save(here, embedder(_features()))
|
||||
|
||||
there = tmp_path / "there.npy"
|
||||
script = (
|
||||
"import types, numpy as np\n"
|
||||
"from portfolio_optimiser.semretrieval import FakeEmbedder\n"
|
||||
"f = types.SimpleNamespace(\n"
|
||||
" affected_codes=frozenset({'05.2', '03.1'}),\n"
|
||||
" measure_type='scope_reduction',\n"
|
||||
" claimed_saving_nok=220000.0,\n"
|
||||
" description='asphalt base course reduction near school',\n"
|
||||
")\n"
|
||||
f"np.save({str(there)!r}, FakeEmbedder()(f))\n"
|
||||
)
|
||||
result = subprocess.run(
|
||||
[sys.executable, "-c", script],
|
||||
env={"PATH": "/usr/bin:/bin", "PYTHONPATH": str(_SRC), "PYTHONHASHSEED": "12345"},
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
assert result.returncode == 0, result.stderr
|
||||
assert here.read_bytes() == there.read_bytes()
|
||||
|
||||
|
||||
def test_cosine_of_identical_unit_vectors_is_one() -> None:
|
||||
"""Approximate by necessity: a float64 dot of a normalized vector with itself lands within
|
||||
an ulp or two of 1.0, never exactly on it. Ranking never depends on the exact value —
|
||||
``HybridRanker`` rounds to 9 decimals and breaks remaining ties on ``id``."""
|
||||
vec = FakeEmbedder()(_features())
|
||||
assert cosine(vec, vec) == pytest.approx(1.0)
|
||||
|
||||
|
||||
def test_cosine_is_bounded_to_unit_interval() -> None:
|
||||
embedder = FakeEmbedder()
|
||||
score = cosine(embedder(_features()), embedder(_features(codes=frozenset({"09.1"}))))
|
||||
assert 0.0 <= score <= 1.0
|
||||
|
||||
|
||||
def test_cosine_of_zero_norm_is_zero() -> None:
|
||||
"""A zero vector must never produce NaN — NaN in a sort key silently corrupts ranking."""
|
||||
zeros = np.zeros(EMBED_DIM, dtype="<f8")
|
||||
vec = FakeEmbedder()(_features())
|
||||
assert cosine(zeros, vec) == 0.0
|
||||
assert cosine(vec, zeros) == 0.0
|
||||
assert cosine(zeros, zeros) == 0.0
|
||||
Loading…
Add table
Add a link
Reference in a new issue