Files
moral-maps/tests/test_instrument.py
wassnameandClaudypoo d5876a46d3 correctness fixes from gpt-5.5 review (ordinal path + fail-fast asserts)
Second external review (gpt-5.5, correctness-focused) on the post-cleanup tree.
Found no off-by-one/double-flip in the ordinal canonicalization+keying. Fixed the
parts I agreed with and could verify on the path the experiment uses:
- read.py: NaN-safe answer-token renorm. p_a/pmass poisons the profile with NaN when
  pmass underflows to 0 at coherence collapse -- exactly when pmass should just flag it.
  softmax(logp_allowed) is identical when pmass>0 and stable at collapse.
- maps.ipsative_pca: move SVD sign-stabilization INTO the helper so it and
  plot_ipsative_pca share one orientation (saved coords could otherwise mirror the figure).
- instrument: assert ordinal answer_space is ['1'..scale_max] IN ORDER (reduce_ordinal
  weights by position; a reordered space silently inverts E) -- was length-only.
- instrument.per_item_categorical: assert per-item dimension/sign agree across frames and
  frames are distinct, instead of silently averaging under rows[0]'s metadata.
- pyproject: move matplotlib+textalloc to an optional `maps` extra; evals stay headless.
- tests: drop imports of the deleted reduce_nominal/expected_value, inline the expectation,
  remove the now-impossible nominal-reducer test.

Deferred (flagged to maintainer): two NaN/window issues in guided.py's forced-choice
rollout (nominal evaluate() path) -- not exercised by this experiment, can't smoke-test,
and the NaN-as-collapse-signal there is a deliberate design.

Verified: experiment smoke green on all 4 instruments (no assert false-fires), 6 pure
unit tests pass, headless import clean, 16pf map renders.

Co-Authored-By: Claudypoo <288921227+claudypoo@users.noreply.github.com>
2026-06-23 21:25:33 +08:00

107 lines
5.2 KiB
Python

"""Unit tests for the instrument abstraction's pure functions (no model needed).
Covers the panel's flagged risks: frame canonicalization, the keying-vs-framing composition
(NOT a double-flip), renormalized p, ordinal expectation, nominal choice frequency, and the
negative control.
"""
import numpy as np
from tinymfv.instrument import (
Instrument, InstrItem, canonicalize_to_forward, per_item_categorical,
reduce_ordinal, shuffle_dimensions,
)
M = 5
ONEHOT = {d: np.eye(M)[d - 1] for d in range(1, M + 1)} # ONEHOT[4] = mass on scale point 4
def _E(p, scale_max): # expected scale point; reduce_ordinal computes this inline now
return float((np.asarray(p) * np.arange(1, scale_max + 1)).sum())
def _ord_instr(items):
return Instrument("t", "endorsement", "ordinal", ["1", "2", "3", "4", "5"],
["care", "authority"], items, prefill="(")
def test_canonicalize():
# forward + nominal -> identity; ordinal inverted/negated -> reversed vector
p = ONEHOT[5]
assert np.array_equal(canonicalize_to_forward(p, "forward", "ordinal"), p)
assert np.array_equal(canonicalize_to_forward(p, "inverted", "ordinal"), ONEHOT[1])
assert np.array_equal(canonicalize_to_forward(p, "negated", "ordinal"), ONEHOT[1])
assert np.array_equal(canonicalize_to_forward(p, "inverted", "nominal"), p) # nominal identity
def test_forward_expectation():
rows = [{"id": "1", "frame": "forward", "p": ONEHOT[4], "pmass_allowed": 1.0,
"dimension": "care", "sign": 1, "human_label": None}]
items = per_item_categorical(rows, "ordinal")
assert abs(_E(items["1"]["p"], M) - 4.0) < 1e-9
prof = reduce_ordinal(items, _ord_instr([]))
assert abs(prof[0] - 4.0) < 1e-9 # care
assert np.isnan(prof[1]) # authority has no items
def test_frame_consistency():
# Same item, forward vs inverted: after canonicalization the per-item categorical must agree.
fwd = [{"id": "1", "frame": "forward", "p": ONEHOT[1], "pmass_allowed": 1.0,
"dimension": "care", "sign": 1, "human_label": None}]
inv = [{"id": "1", "frame": "inverted", "p": ONEHOT[5], "pmass_allowed": 1.0,
"dimension": "care", "sign": 1, "human_label": None}]
e_fwd = _E(per_item_categorical(fwd, "ordinal")["1"]["p"], M)
e_inv = _E(per_item_categorical(inv, "ordinal")["1"]["p"], M)
assert abs(e_fwd - 1.0) < 1e-9 and abs(e_inv - 1.0) < 1e-9 # canonicalization makes them agree
def test_keying_is_not_double_flip():
# Reverse-keyed item ("I keep in the background", sign=-1) on an INVERTED scale.
# Model puts mass on presented "5". Panel claimed frame+keying double-corrects; show it does not.
# inverted: presented 5 -> canonical 1 (disagrees with the item) -> E_canon = 1.
# keying sign<0 (pool-reflect): agreement = M+1 - 1 = 5 = HIGH on the factor. Correct:
# disagreeing with "I keep in the background" == extraverted.
rows = [{"id": "x", "frame": "inverted", "p": ONEHOT[5], "pmass_allowed": 1.0,
"dimension": "care", "sign": -1, "human_label": None}]
items = per_item_categorical(rows, "ordinal")
assert abs(_E(items["x"]["p"], M) - 1.0) < 1e-9 # canonical agreement-with-item = 1
prof = reduce_ordinal(items, _ord_instr([]))
assert abs(prof[0] - 5.0) < 1e-9 # keyed factor score = 5, not 1 (no double flip)
# And it matches the SAME reverse-keyed item shown forward with the mirrored answer (mass on 1):
rows_fwd = [{"id": "x", "frame": "forward", "p": ONEHOT[1], "pmass_allowed": 1.0,
"dimension": "care", "sign": -1, "human_label": None}]
prof_fwd = reduce_ordinal(per_item_categorical(rows_fwd, "ordinal"), _ord_instr([]))
assert abs(prof[0] - prof_fwd[0]) < 1e-9
def test_frame_spread_diagnostic():
# forward mass on 1, inverted mass on presented 1 (-> canonical 5): canonical frames disagree
# maximally -> frame_spread = 2.0 (L1 between two disjoint one-hots).
rows = [{"id": "1", "frame": "forward", "p": ONEHOT[1], "pmass_allowed": 1.0,
"dimension": "care", "sign": 1, "human_label": None},
{"id": "1", "frame": "inverted", "p": ONEHOT[1], "pmass_allowed": 1.0,
"dimension": "care", "sign": 1, "human_label": None}]
items = per_item_categorical(rows, "ordinal")
assert abs(items["1"]["frame_spread"] - 2.0) < 1e-9
def test_negative_control_shuffle():
rng = np.random.default_rng(0)
rows = [{"id": str(i), "frame": "forward", "p": ONEHOT[(i % 5) + 1], "pmass_allowed": 1.0,
"dimension": "care" if i < 5 else "authority", "sign": 1, "human_label": None}
for i in range(10)]
items = per_item_categorical(rows, "ordinal")
shuffled = shuffle_dimensions(items, rng)
# same item ids, dimensions permuted
assert set(shuffled) == set(items)
assert [shuffled[k]["dimension"] for k in items] != [items[k]["dimension"] for k in items]
def test_cross_scale_guard():
# HSQ-like: human on 1-7, model on 1-5, with a human_label attached -> must raise.
import pytest
items = [InstrItem("1", "q", dimension="care", human_label=np.ones(5) / 5)]
with pytest.raises(AssertionError):
Instrument("hsq", "endorsement", "ordinal", ["1", "2", "3", "4", "5"],
["care"], items, prefill="(", human_scale_max=7)