From 6594cc212d045c8f2235975c0e34310566799896 Mon Sep 17 00:00:00 2001
From: wassname <1103714+wassname@users.noreply.github.com>
Date: Thu, 25 Jun 2026 20:22:04 +0800
Subject: [PATCH] api: clarify eval readout names
---
README.md | 24 ++++----
scripts/09_forced_choice.py | 17 +++---
src/tinymfv/__init__.py | 28 +++++-----
src/tinymfv/administer.py | 20 +++----
src/tinymfv/eval.py | 106 +++++++++++++++++++++++-------------
src/tinymfv/guided.py | 33 ++++++-----
src/tinymfv/instrument.py | 20 ++++++-
src/tinymfv/instruments.py | 4 +-
src/tinymfv/readouts.py | 4 +-
9 files changed, 152 insertions(+), 104 deletions(-)
diff --git a/README.md b/README.md
index 90189de..b9e549b 100644
--- a/README.md
+++ b/README.md
@@ -38,12 +38,13 @@ label distribution). `evaluate(model, tok, ...)` runs a forced-choice probe per
condition) and returns a dict with:
- `profile`: mean `p[foundation]` across vignettes, on the same 7-way simplex as the human profile.
-- `top1_acc`, `mean_js`, `mean_nll_T`: agreement vs the human label (None if the config is unlabeled).
-- `mean_pmass_allowed`: the coherence canary -- mean probability mass on valid answer tokens at the
- answer slot. It drops when the model refuses, rambles, or format-collapses, so a degenerate
- intervention is visible independent of which answer it picks.
+- `top1_acc`, `informedness`, `mean_nll_T`: agreement vs the human label (None if the config is unlabeled).
+- `mean_pmass_allowed`: mean full-vocab probability mass on the allowed answer tokens at the answer
+ slot. This is answer-slot format coherence: it drops when the model wants to emit prose, refusal,
+ punctuation, or another out-of-space token, independent of which valid answer is top.
+- `mean_nll_prefill`: mean NLL/token of the forced assistant prefill that leads into the answer slot.
- `per_row` (with `return_per_row=True`): the per-row 7-vec `p`, raw `score` (nats), `pmass_allowed`,
- `top1`, `margin`. This is what the steering metrics below consume.
+ `nll_prefill`, `top1`, `margin`. This is what the steering metrics below consume.
To measure a steering intervention, run `evaluate` twice (base vs steered, same vignettes) and diff
the reports. The steering-lite package wraps this as `evaluate_with_vector(model, tok, vector=v)`,
@@ -83,7 +84,7 @@ more robust: use dlogit for effect size, SI for "did the steer do the intended s
interventions register in nats before changing an argmax.
- Position-bias control. Each row is scored twice (options forward and reversed) and the logprob
vectors averaged, cancelling option-order effects ([Pezeshkpour & Hruschka 2023](https://arxiv.org/abs/2308.11483)).
-- A sliding think budget. `max_think_tokens` (0 / 64 dev default / 4096 / unbounded) is a knob you
+- A sliding think budget. `max_think_tokens` (0 / 64 dev default / 4096 / unbounded) is a setting you
sweep: steering accrues over the think trace, so the same vector moves the profile more with more
think, up to the point (~512) where the model closes `` on its own and the readout collapses.
- Two modes. dev (N=1, greedy, 64 think) is fast and granular, the default. full (N=4 sampled traces
@@ -91,10 +92,13 @@ more robust: use dlogit for effect size, SI for "did the steer do the intended s
## Instruments
-The reader is answer-space-agnostic: it gathers logprobs over a set of answer tokens at a prefilled
-slot (`src/tinymfv/instrument.py`). Forced-choice (nominal, the MFV default) reads a foundation
-choice; Likert (ordinal) reads a 1..M scale point for MFQ-2 / Big-Five / 16PF / humor-styles (spec
-and reducers landed; wiring through `evaluate()` is in progress).
+The reader is answer-space-agnostic: it gathers logprobs over answer tokens at a prefilled slot
+(`src/tinymfv/instrument.py`).
+
+- Nominal instruments, the MFV vignettes, read a foundation category and reduce to mean category
+ probability.
+- Ordinal instruments, MFQ-2 / Big-Five / 16PF / humor-styles, read a 1..M scale point and reduce to
+ keyed expected score `E`, logit contrast `C`, `logodds_agree`, entropy, and `pmass_allowed`.
## Scope
diff --git a/scripts/09_forced_choice.py b/scripts/09_forced_choice.py
index bced4bc..cdfa57a 100644
--- a/scripts/09_forced_choice.py
+++ b/scripts/09_forced_choice.py
@@ -4,7 +4,6 @@ Wraps `tinymfv.evaluate()`. Reports the AI-vs-label distribution match:
top1_acc argmax model == argmax label
mean_nll soft cross-entropy vs human distribution, nats
mean_nll_T same metric after one fitted temperature
- mean_js legacy Jensen-Shannon (model || label), nats; max = ln 2
pearson[f] cross-vignette Pearson(model_p[f], label_p[f]) on
labeled rows (other_violate condition).
@@ -80,10 +79,13 @@ def main() -> None:
"condition": r["condition"],
"foundation_coarse": r["foundation_coarse"],
"p": {f: float(r["p"][i]) for i, f in enumerate(_DEFAULT_FORCED_FOUNDATIONS)},
+ "score": {f: float(r["score"][i]) for i, f in enumerate(_DEFAULT_FORCED_FOUNDATIONS)},
"label": (None if r["label"] is None
else {f: float(r["label"][i]) for i, f in enumerate(_DEFAULT_FORCED_FOUNDATIONS)}),
"top1": r["top1"],
- "margin": float(r["margin"])
+ "margin": float(r["margin"]),
+ "pmass_allowed": float(r["pmass_allowed"]),
+ "nll_prefill": float(r["nll_prefill"]),
}
f.write(json.dumps(rec) + "\n")
logger.info(f"wrote {len(out['per_row'])} rows to {out_path}")
@@ -102,9 +104,8 @@ def main() -> None:
print(f" mean_nll_T = {out['mean_nll_T']} (temperature-scaled, nats)")
print(f" median_nll_T = {out['median_nll_T']} (temperature-scaled, nats)")
print(f" T = {out['T']}")
- print(f" mean_js = {out['mean_js']} (max possible = ln 2 = 0.693)")
print(f" mean_pmass_allowed = {out['mean_pmass_allowed']} (valid-token mass)")
- print(f" mean_nll_json = {out['mean_nll_json']} (assistant prefill, nats/tok)")
+ print(f" mean_nll_prefill = {out['mean_nll_prefill']} (assistant prefill, nats/tok)")
if out["profile"] is not None:
print("\n=== mean profile (human vs model) ===")
@@ -116,13 +117,13 @@ def main() -> None:
f"{np.median(p_top1):.3f} / {p_top1.mean():.3f} / {p_top1.max():.3f}")
print(" SHOULD: median > 0.4 (clear winner per row); <0.2 -> probe broken")
- # JSON-prefill NLL degradation probe (teacher-forced on assistant prefill).
- nll = np.array([float(r["nll_json"]) for r in out["per_row"]])
+ # Prefill NLL degradation probe (teacher-forced on assistant prefill).
+ nll = np.array([float(r["nll_prefill"]) for r in out["per_row"]])
nll = nll[np.isfinite(nll)]
if len(nll):
- print(f"\n nll_json (nats/tok) min/median/mean/max: "
+ print(f"\n nll_prefill (nats/tok) min/median/mean/max: "
f"{nll.min():.3f} / {np.median(nll):.3f} / {nll.mean():.3f} / {nll.max():.3f}")
- print(" SHOULD: stable across runs at fixed model; rises under steering/ablation -> JSON-prefill degradation")
+ print(" SHOULD: stable across runs at fixed model; rises under steering/ablation -> prefill degradation")
if __name__ == "__main__":
diff --git a/src/tinymfv/__init__.py b/src/tinymfv/__init__.py
index 770b2d4..b1372bb 100644
--- a/src/tinymfv/__init__.py
+++ b/src/tinymfv/__init__.py
@@ -1,10 +1,11 @@
-"""tinymfv: tiny moral-foundations vignettes eval.
+"""tinymfv: tiny moral/value instruments for local LLMs.
-Forced-choice 7-way scoring on Clifford 2015 vignettes (classic) +
-paraphrase configs (scifi, ai-actor). Default condition is
-`other_violate` (the canonical Clifford framing); `self_violate` is
-available as an opt-in ablation. Each row internally does a fwd + rev
-enum-order pass for position-bias debias (inside guided_rollout).
+One answer-token reader, two reducer families:
+
+- nominal MFV vignettes: answer = foundation category; `evaluate` reports a
+ 7-way profile plus label-match metrics.
+- ordinal Likert questionnaires: answer = scale point; `administer` reports
+ E, C, agree-vs-disagree log-odds, entropy, and pmass diagnostics.
High-level usage:
@@ -16,17 +17,17 @@ High-level usage:
rep = evaluate(model, tok, name="classic")
print(rep["table"]) # per-foundation
print(rep["top1_acc"]) # argmax accuracy vs label
- print(rep["mean_js"]) # JS divergence vs label dist (in nats)
+ print(rep["mean_nll_T"]) # temperature-scaled soft NLL vs label dist
Lower-level: see `guided_rollout_forced_choice` in `tinymfv.guided`.
"""
from .data import load_vignettes, load_all_vignettes, CONFIGS, ConfigName
-from .eval import evaluate, CONDITIONS
+from .eval import evaluate, CONDITIONS, EvalResult, EvalRow, EvalInfo
from .guided import guided_rollout_forced_choice, _DEFAULT_FORCED_FOUNDATIONS
-from .instrument import Instrument, InstrItem, per_item_categorical
+from .instrument import Instrument, InstrItem, per_item_categorical, reduce_nominal, reduce_ordinal
from .instruments import get as get_instrument, INSTRUMENTS, build_instrument
from .read import read_items, resolve_answer_ids, build_user_content
-from .readouts import expected_score, logit_contrast, agree_logodds, entropy
+from .readouts import expected_score, logit_contrast, logodds_agree, entropy
from .administer import administer
@@ -41,17 +42,16 @@ def __getattr__(name: str):
# The front door. Other symbols above stay importable (e.g. read_items for item subsets,
-# per_item_categorical, build_instrument) but are plumbing, kept out of `import *`.
+# per_item_categorical, build_instrument) but are helper internals, kept out of `import *`.
__all__ = [
# entrypoints
"evaluate", "administer", "get_instrument", "read_items",
# ordinal readouts (pure functions of the raw answer-token logprobs)
- "expected_score", "logit_contrast", "agree_logodds", "entropy",
+ "expected_score", "logit_contrast", "logodds_agree", "entropy",
# types consumers build / subset
- "Instrument", "InstrItem",
+ "Instrument", "InstrItem", "EvalResult", "EvalRow", "EvalInfo", "reduce_nominal", "reduce_ordinal",
# data API
"load_vignettes", "load_all_vignettes", "CONFIGS", "ConfigName", "CONDITIONS",
# lower-level rollout + lazy plotting
"guided_rollout_forced_choice", "maps",
]
-
diff --git a/src/tinymfv/administer.py b/src/tinymfv/administer.py
index 1563cc0..8f1ca46 100644
--- a/src/tinymfv/administer.py
+++ b/src/tinymfv/administer.py
@@ -6,7 +6,7 @@ This is the survey counterpart to `tinymfv.evaluate` (the vignette forced-choice
per_item_categorical (canonicalize each frame to forward, average -> one dist per item)
reduce_ordinal (E[scale point] per item, reverse-key, pool to a per-factor profile)
-The profile vector (per `instr.dimensions`) is the load-bearing output; `per_item_categorical`'s
+The profile vector (per `instr.dimensions`) is the main output; `per_item_categorical`'s
canonicalization makes it algebraically identical to the experiment's `admin.administer` per-factor
means (verified by the reducer parity check), so this is a drop-in for the maps.
@@ -21,7 +21,7 @@ import numpy as np
from .instrument import Instrument, per_item_categorical, reduce_ordinal, canonicalize_to_forward
from .read import read_items, resolve_answer_ids
-from .readouts import expected_score, logit_contrast, agree_logodds, entropy
+from .readouts import expected_score, logit_contrast, logodds_agree, entropy
class ItemRow(TypedDict):
@@ -31,8 +31,8 @@ class ItemRow(TypedDict):
keyed_E: float # E reverse-keyed for sign<0 items (== old keyed_agreement)
C: float # rank-centered logit contrast (primary steer signal)
keyed_C: float # C reverse-keyed (negated) for sign<0 items
- logodds: float # agree-vs-disagree log-odds (readable direction summary)
- keyed_logodds: float
+ logodds_agree: float # agree-vs-disagree log-odds (readable direction summary)
+ keyed_logodds_agree: float
entropy: float # within-allowed entropy, nats (coherence the pmass gate misses)
pmass_allowed: float
frame_spread: float
@@ -60,7 +60,7 @@ class AdministerResult(TypedDict):
profile_C: np.ndarray # [len(dimensions)] per-factor keyed contrast (the steer map input)
profile: np.ndarray # alias of profile_E (kept so existing E maps keep working)
dimensions: list[str] # factor order, matches the profiles
- foundations: list[dict] # one per factor: foundation, mean(E), C, logodds, sd, ci95*, f_
+ foundations: list[dict] # one per factor: foundation, mean(E), C, logodds_agree, sd, ci95*, f_
per_item: list[ItemRow] # one per item, frame-averaged readouts
per_item_frame: list[ItemFrameRow] # one per (item, frame): raw lp + think + readouts
mean_pmass_allowed: float # coherence check (mass on valid answer tokens)
@@ -88,16 +88,16 @@ def administer(model, tok, instr: Instrument, *, batch_size: int = 36,
# per-item frame-averaged readouts. E and entropy come from the averaged probability vector (so the
# NaN-at-collapse signal survives); C and log-odds come from the averaged logprobs (the sensitive
# log-space readouts). Reverse-keying (sign<0): E reflects to M+1-E, while the midpoint-centered
- # contrast C and the log-odds simply negate (reflecting the scale negates a centered weight).
+ # contrast C and the agree-vs-disagree log-odds negate.
per_item_rows = []
for iid, it in items.items():
lp, p, sign = it["lp"], it["p"], it["sign"]
- E, Cval, LO = expected_score(p, M), logit_contrast(lp, M), agree_logodds(lp, M)
+ E, Cval, LO = expected_score(p, M), logit_contrast(lp, M), logodds_agree(lp, M)
per_item_rows.append({
"id": iid, "foundation": it["dimension"],
"E": E, "keyed_E": (M + 1 - E) if sign < 0 else E,
"C": Cval, "keyed_C": -Cval if sign < 0 else Cval,
- "logodds": LO, "keyed_logodds": -LO if sign < 0 else LO,
+ "logodds_agree": LO, "keyed_logodds_agree": -LO if sign < 0 else LO,
"entropy": entropy(p, M), "pmass_allowed": it["pmass"], "frame_spread": it["frame_spread"],
})
@@ -134,7 +134,7 @@ def administer(model, tok, instr: Instrument, *, batch_size: int = 36,
for j, d in enumerate(instr.dimensions):
e_vals = np.array([row["keyed_E"] for row in per_item_rows if row["foundation"] == d])
c_vals = np.array([row["keyed_C"] for row in per_item_rows if row["foundation"] == d])
- lo_vals = np.array([row["keyed_logodds"] for row in per_item_rows if row["foundation"] == d])
+ lo_vals = np.array([row["keyed_logodds_agree"] for row in per_item_rows if row["foundation"] == d])
profile_C[j] = float(np.mean(c_vals))
e_lo, e_hi = _ci(e_vals); c_lo, c_hi = _ci(c_vals)
per_fr = {fr: float(np.mean(by_dim_frame[(d, fr)])) for fr in frames}
@@ -144,7 +144,7 @@ def administer(model, tok, instr: Instrument, *, batch_size: int = 36,
"ci95_lo": e_lo, "ci95_hi": e_hi,
"C": float(profile_C[j]), "C_sd": float(c_vals.std(ddof=1)),
"C_ci95_lo": c_lo, "C_ci95_hi": c_hi,
- "logodds": float(np.mean(lo_vals)),
+ "logodds_agree": float(np.mean(lo_vals)),
"framing_spread": float(max(per_fr.values()) - min(per_fr.values())),
**{f"f_{fr}": v for fr, v in per_fr.items()},
})
diff --git a/src/tinymfv/eval.py b/src/tinymfv/eval.py
index c5aafd5..413ad55 100644
--- a/src/tinymfv/eval.py
+++ b/src/tinymfv/eval.py
@@ -22,16 +22,13 @@ Headline metrics:
steering-lite's surgical informedness.
- mean_nll: mean soft cross-entropy -sum_f p_human[f] log p_model[f], in nats.
- mean_nll_T: same metric after fitting one temperature on the scored set.
-- mean_js: legacy mean Jensen-Shannon divergence between model and label dist
- (in nats, max = ln 2 ≈ 0.693).
- gap[f]: per-foundation perspective gap = mean p[f] (other_violate)
- mean p[f] (self_violate). Detects perspective bias.
"""
from __future__ import annotations
import json
-import math
import time
-from typing import Any
+from typing import Any, NotRequired, TypedDict
import numpy as np
import pandas as pd
@@ -58,6 +55,56 @@ _PROBE_TO_COARSE: dict[str, str] = {
_COARSE_NORM = {"Social Norms": "SocialNorms"}
+class EvalRow(TypedDict):
+ id: str
+ condition: str
+ foundation_coarse: str
+ p: np.ndarray # answer distribution, renormalized over the allowed answer space
+ score: np.ndarray # debiased pre-softmax answer evidence, nats
+ label: np.ndarray | None # human answer distribution in the same order as p
+ top1: str
+ margin: float # score[top1] - score[top2], nats
+ pmass_allowed: float # full-vocab mass on the allowed answer tokens at the answer slot
+ nll_prefill: float # NLL/token of the forced assistant prefill before the answer slot
+ think_tokens: list[int]
+ think_tokens_rev: list[int]
+ emitted_close: list[bool]
+ emitted_close_rev: list[bool]
+ gen_text: list[str]
+ gen_text_rev: list[str]
+ lp_fwd_samples: list[list[float]]
+ lp_rev_samples: list[list[float]]
+
+
+class EvalInfo(TypedDict):
+ name: str
+ n_rows: int
+ n_labeled: int
+ elapsed_s: float
+ mean_nll: float | None
+ median_nll: float | None
+ median_nll_T: float | None
+ informedness: float | None
+ mean_pmass_allowed: float | None
+ mean_nll_prefill: float | None
+
+
+class EvalResult(TypedDict):
+ table: pd.DataFrame
+ profile: pd.DataFrame | None
+ mean_nll: float | None
+ mean_nll_T: float | None
+ median_nll_T: float | None
+ T: float | None
+ top1_acc: float | None
+ informedness: float | None
+ mean_pmass_allowed: float | None
+ mean_nll_prefill: float | None
+ info: EvalInfo
+ demos: dict[str, Any] | None
+ per_row: NotRequired[list[EvalRow]]
+
+
def _label_dist(row: dict, foundations: list[str]) -> np.ndarray | None:
"""Build the 7-vec human label distribution for a vignette.
@@ -78,16 +125,6 @@ def _label_dist(row: dict, foundations: list[str]) -> np.ndarray | None:
return arr / s
-def _js_divergence(p: np.ndarray, q: np.ndarray) -> float:
- """Jensen-Shannon divergence in nats. Symmetric, bounded by ln 2."""
- p = p + 1e-12; q = q + 1e-12
- p = p / p.sum(); q = q / q.sum()
- m = 0.5 * (p + q)
- kl_pm = float((p * np.log(p / m)).sum())
- kl_qm = float((q * np.log(q / m)).sum())
- return 0.5 * kl_pm + 0.5 * kl_qm
-
-
def _soft_nll(p_human: np.ndarray, p_model: np.ndarray) -> float:
"""Soft cross-entropy: -sum_f p_human[f] log p_model[f], in nats.
@@ -178,7 +215,7 @@ def evaluate(
device: str | None = None,
return_per_row: bool = False,
verbose: int = 1,
-) -> dict[str, Any]:
+) -> EvalResult:
"""Run forced-choice 7-way probe per (vignette, condition).
Args:
@@ -212,11 +249,11 @@ def evaluate(
table, and the complete DEMO B (prompt + generation + SHOULD note).
Returns:
- Dict with `table`, `profile`, `mean_js`, `mean_nll`, `mean_nll_T`,
- `median_nll_T`, `T`, `top1_acc`, `mean_pmass_allowed`, `mean_nll_json`, and `info`.
+ Dict with `table`, `profile`, `mean_nll`, `mean_nll_T`,
+ `median_nll_T`, `T`, `top1_acc`, `mean_pmass_allowed`, `mean_nll_prefill`, and `info`.
With `return_per_row=True`, also includes `per_row` with per-row
`p`, `score` (debiased logp per foundation), `pmass_allowed`,
- `nll_json`, `gen_text` / `gen_text_rev` (full decoded gen, no stripping),
+ `nll_prefill`, `gen_text` / `gen_text_rev` (full decoded gen, no stripping),
and `top1` / `margin`.
"""
if vignettes is None:
@@ -233,7 +270,7 @@ def evaluate(
foundations = list(_DEFAULT_FORCED_FOUNDATIONS)
t0 = time.time()
- per_row: list[dict] = []
+ per_row: list[EvalRow] = []
total_calls = len(vignettes) * len(conditions)
with tqdm(total=total_calls, desc=f"forced-choice {name}", mininterval=60, maxinterval=120) as pbar:
for cond in conditions:
@@ -265,7 +302,7 @@ def evaluate(
"top1": res.top1,
"margin": res.margin,
"pmass_allowed": res.pmass_allowed,
- "nll_json": res.nll_json,
+ "nll_prefill": res.nll_prefill,
"think_tokens": res.think_tokens, # list[int], length N
"think_tokens_rev": res.think_tokens_rev, # list[int], length N
"emitted_close": res.emitted_close, # list[bool], length N
@@ -332,9 +369,6 @@ def evaluate(
# === headline scalars (need labels) ===
labeled_rows = [r for r in per_row if r["label"] is not None]
if labeled_rows:
- js_vals = np.array([_js_divergence(r["p"], r["label"]) for r in labeled_rows])
- mean_js = float(js_vals.mean())
- median_js = float(np.median(js_vals))
y_pred = np.array([np.argmax(r["p"]) for r in labeled_rows])
y_true = np.array([np.argmax(r["label"]) for r in labeled_rows])
top1_acc = float(np.mean(y_pred == y_true))
@@ -371,7 +405,7 @@ def evaluate(
"model_T": p_scaled.mean(axis=0),
})
else:
- mean_js = median_js = top1_acc = informedness = None
+ top1_acc = informedness = None
mean_nll = median_nll = mean_nll_T = median_nll_T = None
T = None
profile = None
@@ -380,8 +414,8 @@ def evaluate(
float(np.mean([r["pmass_allowed"] for r in per_row]))
if per_row else None
)
- mean_nll_json = (
- float(np.mean([r["nll_json"] for r in per_row]))
+ mean_nll_prefill = (
+ float(np.mean([r["nll_prefill"] for r in per_row]))
if per_row else None
)
@@ -394,7 +428,7 @@ def evaluate(
r0 = per_row[0]
# one-line quantitative readout, kept at every verbose level (it IS the signal)
aux = {k: (round(v, 4) if isinstance(v, float) else v) for k, v in {
- "top1_acc": top1_acc, "mean_js": mean_js, "mean_nll_T": mean_nll_T,
+ "top1_acc": top1_acc, "mean_nll_T": mean_nll_T,
"T": T, "informedness": informedness, "mean_pmass_allowed": mean_pmass_allowed,
}.items() if v is not None}
logger.debug("aux stats: " + json.dumps(aux))
@@ -407,7 +441,7 @@ def evaluate(
"SHOULD: mass concentrates on the violated foundation; if it is flat or "
"pmass_allowed~0 the model did not answer in-format and the row is noise.\n"
+ " ".join(f"{f}={p:.3f}" for f, p in zip(foundations, r0["p"]))
- + f"\n top1={r0['top1']} pmass_allowed={r0['pmass_allowed']:.3f} nll_json={r0['nll_json']:.3f}"
+ + f"\n top1={r0['top1']} pmass_allowed={r0['pmass_allowed']:.3f} nll_prefill={r0['nll_prefill']:.3f}"
)
if profile is not None:
logger.debug(
@@ -432,8 +466,8 @@ def evaluate(
f"{demo_prompt}{demo_gen}\n"
"SHOULD: a real chain-of-thought that ends in a moral-foundation choice. "
"If it is empty or degenerate the model is not reasoning at this budget; "
- "if it answers a different foundation than DEMO A's top1, the readout and "
- "free reasoning disagree (worth noting).\n--- end DEMO B ---\n"
+ "if it answers a different foundation than DEMO A's top1, inspect that row.\n"
+ "--- end DEMO B ---\n"
)
else: # terse default: generation only, whitespace-collapsed to 64 chars, bracketed
gen64 = " ".join(demo_gen.split())[:64]
@@ -451,8 +485,6 @@ def evaluate(
"n_rows": n_rows,
"n_labeled": n_labeled,
"elapsed_s": elapsed,
- "median_js": median_js,
- "max_js": math.log(2),
"mean_nll": mean_nll,
"median_nll": median_nll,
"median_nll_T": median_nll_T,
@@ -468,14 +500,13 @@ def evaluate(
# "in-format"; a sharp drop after steering signals coherence loss.
"mean_pmass_allowed": mean_pmass_allowed,
# Mean NLL in nats/token over the assistant prefill content. Perplexity
- # is exp(mean_nll_json).
- "mean_nll_json": mean_nll_json,
+ # is exp(mean_nll_prefill).
+ "mean_nll_prefill": mean_nll_prefill,
}
- out: dict[str, Any] = {
+ out: EvalResult = {
"table": table,
"profile": profile, # 7-row DataFrame: foundation, human, model, model_T
- "mean_js": mean_js,
"mean_nll": mean_nll,
"mean_nll_T": mean_nll_T, # temperature-scaled soft cross-entropy in nats
"median_nll_T": median_nll_T,
@@ -483,11 +514,10 @@ def evaluate(
"top1_acc": top1_acc,
"informedness": informedness, # macro Youden's J, model vs human argmax, in [-1, 1]
"mean_pmass_allowed": mean_pmass_allowed,
- "mean_nll_json": mean_nll_json,
+ "mean_nll_prefill": mean_nll_prefill,
"info": info,
"demos": demos, # DEMO A (forced think + top1) + DEMO B (free reasoning); None if not verbose
}
if return_per_row:
out["per_row"] = per_row
return out
-
diff --git a/src/tinymfv/guided.py b/src/tinymfv/guided.py
index 6dfe0e8..662b8e0 100644
--- a/src/tinymfv/guided.py
+++ b/src/tinymfv/guided.py
@@ -108,7 +108,7 @@ def _rollout_natural_or_forced(
callers reshape via `[i*N + n]`.
thinks[j] = (gen_text, n_think_tokens, emitted_close).
- slots[j][k] = {pmass_allowed, nll_json, top5_str, lp_gather}.
+ slots[j][k] = {pmass_allowed, nll_prefill, top5_str, lp_gather}.
Phase 1: batched generate, `min_new_tokens=max_new_tokens=max_think_tokens`
→ uniform-length cache. Capture `scores` (per-step logits) and `pkv`.
@@ -236,12 +236,12 @@ def _rollout_natural_or_forced(
first_logp = F.log_softmax(prefix_out.logits[:, -1].float(), dim=-1) # [B, V]
first_nll = -first_logp.gather(1, prefill_t[:, :1]).squeeze(-1) # [B]
if J == 1:
- forced_nll_json = first_nll
+ forced_nll_prefill = first_nll
else:
next_logp = F.log_softmax(prefill_out.logits[:, :-1].float(), dim=-1) # [B, J-1, V]
next_ids = prefill_t[:, 1:].unsqueeze(-1) # [B, J-1, 1]
tail_nll = -next_logp.gather(2, next_ids).squeeze(-1).sum(dim=1) # [B]
- forced_nll_json = (first_nll + tail_nll) / J
+ forced_nll_prefill = (first_nll + tail_nll) / J
if verbose:
real0 = phase1_ids[0][phase1_ids[0] != pad_id]
@@ -282,7 +282,7 @@ def _rollout_natural_or_forced(
if not torch.isfinite(raw).all():
slots[i].append({
"pmass_allowed": 0.0,
- "nll_json": float("nan"),
+ "nll_prefill": float("nan"),
"top5_str": "",
"lp_gather": [float("nan")] * len(gather_token_ids),
})
@@ -297,18 +297,18 @@ def _rollout_natural_or_forced(
elif not emitted_close_i:
# Case (b) interrupted: forced
lp_vec = forced_lp_last[i]
- nll_val = float(forced_nll_json[i].item())
+ nll_val = float(forced_nll_prefill[i].item())
else:
# Case (c) emitted but no natural answer slot found.
# Model "finished thinking" without producing JSON — coherence
# collapse at the answer slot. pmass=0.0 is the honest measurement
# (no probability mass on allowed tokens at a non-existent slot)
# and lets c_scan see the failure as a real signal rather than
- # crashing on NaN. nll_json stays NaN (genuinely undefined: no
- # JSON tokens were emitted to score).
+ # crashing on NaN. nll_prefill stays NaN (genuinely undefined:
+ # no prefill tokens were emitted to score).
slots[i].append({
"pmass_allowed": 0.0,
- "nll_json": float("nan"),
+ "nll_prefill": float("nan"),
"top5_str": "",
"lp_gather": [float("nan")] * len(gather_token_ids),
})
@@ -321,7 +321,7 @@ def _rollout_natural_or_forced(
)
slots[i].append({
"pmass_allowed": float(lp_vec[gid_t].exp().sum().item()),
- "nll_json": nll_val,
+ "nll_prefill": nll_val,
"top5_str": top5_str,
"lp_gather": lp_vec[gid_t].cpu().tolist(),
})
@@ -442,8 +442,8 @@ class ForcedChoiceResult:
pmass_allowed: float
# Mean negative log-likelihood in nats/token over the assistant prefill
# content, averaged across samples and fwd + rev framings. Perplexity is
- # `exp(nll_json)`.
- nll_json: float
+ # `exp(nll_prefill)`.
+ nll_prefill: float
def _resolve_first_token_ids(tok, words: list[str]) -> tuple[list[int], dict[str, int]]:
@@ -599,14 +599,14 @@ def guided_rollout_forced_choice(
order_sorted = sorted(range(K), key=lambda k: -score[k])
top1 = foundations[order_sorted[0]]
margin = score[order_sorted[0]] - score[order_sorted[1]]
- # Average pmass_allowed and nll_json across N samples per direction, then across
+ # Average pmass_allowed and nll_prefill across N samples per direction, then across
# fwd + rev framings.
pm_f = sum(slots_fwd[j][0]["pmass_allowed"] for j in idx) / N
pm_r = sum(slots_rev[j][0]["pmass_allowed"] for j in idx) / N
pm = 0.5 * (pm_f + pm_r)
- nll_f = sum(slots_fwd[j][0]["nll_json"] for j in idx) / N
- nll_r = sum(slots_rev[j][0]["nll_json"] for j in idx) / N
- nll_json = 0.5 * (nll_f + nll_r)
+ nll_f = sum(slots_fwd[j][0]["nll_prefill"] for j in idx) / N
+ nll_r = sum(slots_rev[j][0]["nll_prefill"] for j in idx) / N
+ nll_prefill = 0.5 * (nll_f + nll_r)
results.append(ForcedChoiceResult(
user_prompt=user_prompts[i],
gen_text=gens_fwd,
@@ -624,7 +624,7 @@ def guided_rollout_forced_choice(
emitted_close=close_fwd_list,
emitted_close_rev=close_rev_list,
pmass_allowed=float(pm),
- nll_json=float(nll_json),
+ nll_prefill=float(nll_prefill),
))
return results
@@ -665,4 +665,3 @@ def free_generation_demo(
out = model.generate(**enc, **gen_kwargs)
gen_text = tok.decode(out[0, enc.input_ids.shape[1]:], skip_special_tokens=False)
return prompt_text, gen_text
-
diff --git a/src/tinymfv/instrument.py b/src/tinymfv/instrument.py
index b18e002..6ec23df 100644
--- a/src/tinymfv/instrument.py
+++ b/src/tinymfv/instrument.py
@@ -22,7 +22,7 @@ asymmetric and risks double-flipping. Both kinds unify on one rule:
ordinal `inverted`/`negated` frames reverse the probability vector (the distribution-level
analog of agreement = M+1 - E). The per-item object is then the mean of these canonical
distributions over frames -- a single forward-orientation categorical, used IDENTICALLY for:
- - metrics: soft-NLL / JS / temperature T vs the forward human histogram (both kinds), plus
+ - metrics: soft-NLL / temperature T vs the forward human histogram (both kinds), plus
an ORDINAL metric (mean |E_model - E_human|) so the certifying metric is sensitive to the
expectation the profile actually uses; top1 / informedness only for nominal (an argmax
flip metric is meaningless on an ordered scale: 4-vs-5 != 1-vs-5).
@@ -79,7 +79,7 @@ class Instrument:
f"got {self.answer_space}")
if self.kind == "nominal" and self.answer_to_dim is None:
self.answer_to_dim = {a: a for a in self.answer_space}
- # Cross-scale caveat: a 1-7 human histogram (HSQ) cannot share a 5-way soft-NLL/JS
+ # Cross-scale caveat: a 1-7 human histogram (HSQ) cannot share a 5-way soft-NLL
# with a 1-5 model directly. Calibration for such
# instruments must project both to a common support (or report 0-1 endorsement only).
# Enforced loudly rather than silently mis-comparing:
@@ -155,7 +155,21 @@ def per_item_categorical(per_row: list[dict], kind: Kind) -> dict[str, dict]:
return out
-# --- profile reducer: the ordinal kind-specific summary (nominal evaluate folds its profile inline) ---
+# --- profile reducers: same per-item categorical, different profile summary ---
+
+def reduce_nominal(items: dict[str, dict], instr: Instrument) -> np.ndarray:
+ """Nominal profile = mean category probability per dimension.
+
+ For MFV, the answer is the foundation. The reducer maps answer tokens into profile dimensions
+ and averages the canonical per-item categorical distributions.
+ """
+ assert instr.kind == "nominal", "reduce_nominal expects a nominal instrument"
+ assert instr.answer_to_dim is not None, f"{instr.name}: nominal instrument needs answer_to_dim"
+ by_dim: dict[str, list[float]] = {d: [] for d in instr.dimensions}
+ for it in items.values():
+ for answer, p in zip(instr.answer_space, it["p"]):
+ by_dim[instr.answer_to_dim[answer]].append(float(p))
+ return np.array([float(np.mean(by_dim[d])) for d in instr.dimensions])
def reduce_ordinal(items: dict[str, dict], instr: Instrument) -> np.ndarray:
"""Likert profile = mean keyed agreement per dimension; agreement = E[scale point].
diff --git a/src/tinymfv/instruments.py b/src/tinymfv/instruments.py
index 228ba39..997fb6d 100644
--- a/src/tinymfv/instruments.py
+++ b/src/tinymfv/instruments.py
@@ -42,10 +42,10 @@ _SPECS = {
}
-def _load_keying(survey_dir: Path, fallback_ids: list[str]) -> dict[str, int]:
+def _load_keying(survey_dir: Path, default_forward_ids: list[str]) -> dict[str, int]:
p = survey_dir / "keying.json"
if not p.exists():
- return {i: 1 for i in fallback_ids} # MFQ-2: no file -> every item +1
+ return {i: 1 for i in default_forward_ids} # MFQ-2: every item is +1
return {str(k): int(v) for k, v in json.loads(p.read_text()).items()}
diff --git a/src/tinymfv/readouts.py b/src/tinymfv/readouts.py
index 66b37ad..df3755a 100644
--- a/src/tinymfv/readouts.py
+++ b/src/tinymfv/readouts.py
@@ -5,7 +5,7 @@ matters because the summaries answer different questions and have very different
expected_score E = sum_k k * p_k, in [1, M] human-comparable scale score
logit_contrast C = sum_k (k - mid) * lp_k primary steer signal (sensitive, signed)
- agree_logodds LO = lse(top) - lse(bottom) readable 2-bin direction summary
+ logodds_agree LO = lse(top) - lse(bottom) readable 2-bin direction summary
entropy H = -sum_k p_k log p_k within-allowed coherence (uniform = ln M)
Why E hides steering and C/LO do not (the whole reason this module exists):
@@ -51,7 +51,7 @@ def logit_contrast(lp: np.ndarray, scale_max: int) -> float:
return float((np.asarray(lp, dtype=float) * w).sum())
-def agree_logodds(lp: np.ndarray, scale_max: int) -> float:
+def logodds_agree(lp: np.ndarray, scale_max: int) -> float:
"""LO = logsumexp(top n_side) - logsumexp(bottom n_side), n_side = scale_max // 2.
The readable 2-bin direction summary: nats in favor of agreeing over disagreeing among the