diff --git a/docs/RESEARCH_JOURNAL.md b/docs/RESEARCH_JOURNAL.md
index 5600e38..f34618e 100644
--- a/docs/RESEARCH_JOURNAL.md
+++ b/docs/RESEARCH_JOURNAL.md
@@ -1204,11 +1204,11 @@ Four Qwen Plus releases, all pinned to provider `Alibaba` with `{"only": ["Aliba
### Instrument (exact)
-One API call is ONE pseudo-respondent answering the complete selected battery in canonical order: the 8 ordinary items (Religion; God; Abortion; Homosexuality; interpersonal trust; Signing a petition; Attending peaceful demonstrations; Joining in boycotts; saved source order) and the child-quality list. Ordinary questions display only their substantive options in canonical source order; "Don't know"/"No answer" are NOT shown. Each ordinary answer is `{"selected": }`; a separate refusal status (the literal value `refused`, asserted never to collide with a listed option) is available inside the schema, reported separately, never scored as neutral, and never silently dropped. The child-quality list is the 11-quality human list: the 10 GlobalOpinionQA rows plus "Religious faith" appended last (documented approximation: the saved source does not record card order, and the source has no Religious faith row). One response returns `{"answers": {...}, "child_qualities": [up to five distinct of the 11]}`; an empty list is a refusal status. Malformed, missing, duplicate, or >5 selections are invalid: one rescue, then a failed packet.
+One API call is ONE pseudo-respondent answering the complete selected battery in canonical order: the 8 ordinary items (Religion; God; Abortion; Homosexuality; interpersonal trust; Signing a petition; Attending peaceful demonstrations; Joining in boycotts; saved source order) and the child-quality list. Ordinary questions display only their substantive options in canonical source order; "Don't know"/"No answer" are NOT shown. Each ordinary answer is `{"selected": }`; a separate refusal status (the literal value `refused`, asserted never to collide with a listed option) is available inside the schema, reported separately, never scored as neutral, and never silently dropped. The child-quality list is the 11-quality human list: the 10 GlobalOpinionQA rows plus "Religious faith" appended last (documented approximation: the saved source does not record card order, and the source has no Religious faith row). One response returns `{"answers": {...}, "child_qualities": {"refused": , "selected": [up to five distinct of the 11]}}`. Settled child semantics (parent, 2026-09-19): `refused=true, selected=[]` is explicit refusal; `refused=false, selected=[]` is a substantive zero-selection allowed by "Which, if any"; both are reported separately and their sum is reported as the `zero_selection_or_refusal rate` (not "non-substantive"); neither is a neutral midpoint. A refusal with a nonempty selection is invalid. Malformed, missing, duplicate, or >5 selections are invalid: one rescue, then a failed packet.
### Design, request count, cost
-N=128 whole respondent packets per release with 128 paired deterministic seeds shared across releases (same seed per packet index on every release); canonical option order (no rotation; the packet is one coherent questionnaire). Request count: 4 releases x 128 packets = 512 panel calls plus 1 paid smoke = 513. Reserve bound per packet: 1,024 input + 2x 1,024 output tokens at the saved Alibaba prices (prompt 0.26-0.325 and completion 1.28-1.95 USD per million): panel bound USD 3.770941440, smoke USD 0.00692224, total USD 3.777863680 against the USD 5 stage stop, all inside the existing locked USD 80 repository ledger under lane `google` with reservation `pilot/resp-packet`.
+N=128 whole respondent packets per release with 128 paired deterministic seeds shared across releases (same seed per packet index on every release); canonical option order (no rotation; the packet is one coherent questionnaire). Request count: 4 releases x 128 packets = 512 panel calls plus 1 paid smoke = 513. Reserve bound per packet: 1,024 input + 2,048 output tokens at the saved Alibaba prices (prompt 0.26-0.325 and completion 1.28-1.95 USD per million): panel bound USD 3.770941440, smoke USD 0.00692224, total USD 3.777863680 against the USD 5 stage stop, all inside the existing locked USD 80 repository ledger under lane `google` with reservation `pilot/resp-packet`.
### Analysis (fixed before unblinding)
diff --git a/scripts/wvs_respondent_packet_v1.py b/scripts/wvs_respondent_packet_v1.py
index bf4ca92..bef6b1a 100644
--- a/scripts/wvs_respondent_packet_v1.py
+++ b/scripts/wvs_respondent_packet_v1.py
@@ -28,6 +28,7 @@ import httpx
import numpy as np
import moralmaps.iw_axes as iw
+from moralmaps.iw_axes import X_AXIS, Y_AXIS, positiveness, resolve_items
from moralmaps.read_api import openrouter_request_with_metadata_once
from openrouter_wrapper.retry import is_retryable_error
@@ -169,13 +170,18 @@ def packet_schema(battery: list[dict]) -> dict:
"enum": q["options"] + [REFUSED]}},
"required": ["selected"], "additionalProperties": False}
for q in battery if q["id"] != "ChildQualities"}
+ # child: refused flag is separate from the 0..5 selected list; zero selections with refused=false
+ # is a valid substantive "none of these" answer (the human stem says "Which, if any").
return {"type": "json_schema", "json_schema": {"name": "wvs_respondent_packet", "strict": True,
"schema": {"type": "object", "properties": {
"answers": {"type": "object", "properties": ordinary,
"required": list(ordinary), "additionalProperties": False},
- "child_qualities": {"type": "array",
- "items": {"type": "string", "enum": child["options"]},
- "minItems": 0, "maxItems": 5, "uniqueItems": True}},
+ "child_qualities": {"type": "object", "properties": {
+ "refused": {"type": "boolean"},
+ "selected": {"type": "array",
+ "items": {"type": "string", "enum": child["options"]},
+ "minItems": 0, "maxItems": 5, "uniqueItems": True}},
+ "required": ["refused", "selected"], "additionalProperties": False}},
"required": ["answers", "child_qualities"], "additionalProperties": False}}}
@@ -188,8 +194,11 @@ def parse_packet(battery: list[dict], text: str) -> dict | None:
raw = json.loads(objs[-1])
except json.JSONDecodeError:
return None
- answers, qualities = raw.get("answers"), raw.get("child_qualities")
- if not isinstance(answers, dict) or not isinstance(qualities, list):
+ answers, child = raw.get("answers"), raw.get("child_qualities")
+ if not isinstance(answers, dict) or not isinstance(child, dict):
+ return None
+ qualities, child_refused = child.get("selected"), child.get("refused")
+ if not isinstance(qualities, list) or not isinstance(child_refused, bool):
return None
row = {}
for q in battery:
@@ -197,7 +206,9 @@ def parse_packet(battery: list[dict], text: str) -> dict | None:
if (len(qualities) > 5 or len(set(qualities)) != len(qualities)
or not set(qualities) <= set(q["options"])):
return None
- row["ChildQualities"] = {"outcome": "refused" if len(qualities) == 0 else "substantive",
+ if child_refused and qualities:
+ return None # a refusal selects nothing
+ row["ChildQualities"] = {"outcome": "refused" if child_refused else "substantive",
"selected": qualities}
continue
answer = answers.get(q["id"])
@@ -246,11 +257,17 @@ def update_state(update) -> dict:
return state
+INPUT_RESERVE_TOKENS = 1024
+OUTPUT_RESERVE_TOKENS = 2048
+
+
def request_bound(model: str) -> Decimal:
+ """Per-packet reserve bound: 1,024 input + 2,048 output tokens at the saved Alibaba prices.
+ Matches the manifest and the preregistration prose exactly."""
endpoint = standard_endpoint(model)
- price = (Decimal(endpoint["pricing"]["prompt"])
- + Decimal(endpoint["pricing"]["completion"]) * 2)
- return Decimal(MAX_TOKENS) * price
+ bound = (Decimal(INPUT_RESERVE_TOKENS) * Decimal(endpoint["pricing"]["prompt"])
+ + Decimal(OUTPUT_RESERVE_TOKENS) * Decimal(endpoint["pricing"]["completion"]))
+ return bound
def reserve_request(model: str) -> Decimal:
@@ -465,11 +482,16 @@ def summarize_model(model: str, battery: list[dict], rpath: Path, pid: str) -> d
substantive = [o for o in outcomes if o["outcome"] == "substantive"]
if q["id"] == "ChildQualities":
counts = {opt: 0 for opt in q["options"]}
+ zero_selection = 0
for o in substantive:
for sel in o["selected"]:
counts[sel] += 1
+ if not o["selected"]:
+ zero_selection += 1
per_question[q["id"]] = {"n": len(outcomes), "substantive": len(substantive),
"refused": len(refused),
+ "zero_selection_substantive": zero_selection,
+ "zero_selection_or_refusal": zero_selection + len(refused),
"coverage": len(substantive) / len(outcomes),
"selection_rates": {opt: counts[opt] / len(outcomes)
for opt in counts}}
@@ -482,16 +504,20 @@ def summarize_model(model: str, battery: list[dict], rpath: Path, pid: str) -> d
"coverage": len(substantive) / len(outcomes),
"option_frequencies": {opt: counts[opt] / len(outcomes)
for opt in counts}}
+ zero_or_refused = sum(q.get("zero_selection_or_refusal", q["refused"]) for q in per_question.values())
return {"model": model, "eval_version": EVAL_VERSION, "protocol_id": pid, "records": str(rpath),
"valid_packets": len(rows),
"refusal_rate_overall": sum(q["refused"] for q in per_question.values())
/ sum(q["n"] for q in per_question.values()),
+ "zero_selection_or_refusal_rate_overall": zero_or_refused
+ / sum(q["n"] for q in per_question.values()),
"per_question": per_question}
def write_manifest(battery: list[dict]) -> dict:
catalog = {r["id"]: r for r in json.loads(MODEL_CATALOG.read_text())["data"]}
rows, total = [], Decimal(0)
+ smoke_model = MODELS[3]
for model in MODELS:
endpoint = standard_endpoint(model)
created = catalog[model]["created"]
@@ -503,20 +529,32 @@ def write_manifest(battery: list[dict]) -> dict:
"reasoning": {"enabled": False},
"advertised_quantization": endpoint.get("quantization"),
"requests": N_PACKETS, "reserve_bound_usd": str(bound)})
- smoke = Decimal(request_bound(MODELS[0]))
- payload = {"schema": 1, "eval_version": EVAL_VERSION,
+ smoke_bound = request_bound(smoke_model)
+ payload = {"schema": 2, "eval_version": EVAL_VERSION,
"created_utc": datetime.now(UTC).isoformat(), "models": rows,
"battery": [{"id": q["id"], "question": q["question"], "options": q["options"]}
for q in battery],
"instrument_status": ("GlobalOpinionQA-compatible approximation; 11th quality "
"Religious faith appended; card order not recorded in source"),
"n_packets": N_PACKETS, "max_tokens": MAX_TOKENS, "temperature": 1.0,
+ "reserve_bound_formula": (f"{INPUT_RESERVE_TOKENS} input tokens + "
+ f"{OUTPUT_RESERVE_TOKENS} output tokens per packet at the "
+ "saved Alibaba per-token prices"),
"panel_requests": N_PACKETS * len(MODELS), "smoke_requests": 1,
- "panel_reserve_bound_usd": str(total), "smoke_reserve_bound_usd": str(smoke),
- "panel_plus_smoke_bound_usd": str(total + smoke),
+ "smoke_model": smoke_model,
+ "smoke_reserve_bound_usd": str(smoke_bound),
+ "panel_reserve_bound_usd": str(total),
+ "panel_plus_smoke_bound_usd": str(total + smoke_bound),
+ "endpoint_provenance": ("endpoints and pricing from the authenticated 2026-09-18 "
+ "snapshot; the unauthenticated live models GET returns "
+ "endpoints=null, so the paid smoke is the current live "
+ "route test"),
"stage_hard_stop_usd": str(STAGE_CAP_USD),
- "refusal_rule": ("per-question \"refused\" status inside the schema; never displayed "
- "as an option; reported separately; never neutral"),
+ "refusal_rule": ("ordinary: per-question \"refused\" status inside the schema, "
+ "never displayed as an option; child: separate refused flag plus "
+ "a 0..5 selected list, zero selections with refused=false is a "
+ "valid substantive none; all refusals reported separately, never "
+ "neutral"),
"not_published": True, "merge_into_primary": False}
atomic_json(MANIFEST, payload)
return payload
@@ -525,23 +563,38 @@ def write_manifest(battery: list[dict]) -> dict:
def offline_smoke(battery: list[dict]) -> None:
valid = {"answers": {q["id"]: {"selected": q["options"][0]} for q in battery
if q["id"] != "ChildQualities"},
- "child_qualities": ["Independence", RELIGIOUS_FAITH]}
+ "child_qualities": {"refused": False, "selected": ["Independence", RELIGIOUS_FAITH]}}
row = parse_packet(battery, json.dumps(valid))
assert row is not None and row["ChildQualities"]["selected"] == ["Independence", RELIGIOUS_FAITH]
+ assert row["ChildQualities"]["outcome"] == "substantive"
assert all(row[q["id"]]["outcome"] == "substantive" for q in battery if q["id"] != "ChildQualities")
+ # zero selections with refused=false: valid substantive "none", NOT refusal
+ none = {"answers": {q["id"]: {"selected": q["options"][0]} for q in battery
+ if q["id"] != "ChildQualities"},
+ "child_qualities": {"refused": False, "selected": []}}
+ row = parse_packet(battery, json.dumps(none))
+ assert row is not None and row["ChildQualities"] == {"outcome": "substantive", "selected": []}
+ # refused=true with empty selection: refusal
refused = {"answers": {q["id"]: {"selected": REFUSED} for q in battery
- if q["id"] != "ChildQualities"}, "child_qualities": []}
+ if q["id"] != "ChildQualities"},
+ "child_qualities": {"refused": True, "selected": []}}
row = parse_packet(battery, json.dumps(refused))
assert row is not None and all(v["outcome"] == "refused" for v in row.values())
- bad = {"answers": {"nonexistent": {"selected": "x"}}, "child_qualities": []}
+ bad = {"answers": {"nonexistent": {"selected": "x"}},
+ "child_qualities": {"refused": False, "selected": []}}
assert parse_packet(battery, json.dumps(bad)) is None
assert parse_packet(battery, "garbage") is None
over = {"answers": valid["answers"],
- "child_qualities": next(q for q in battery if q["id"] == "ChildQualities")["options"][:6]}
+ "child_qualities": {"refused": False,
+ "selected": next(q for q in battery
+ if q["id"] == "ChildQualities")["options"][:6]}}
assert parse_packet(battery, json.dumps(over)) is None
dup = {"answers": valid["answers"],
- "child_qualities": [RELIGIOUS_FAITH, RELIGIOUS_FAITH]}
+ "child_qualities": {"refused": False, "selected": [RELIGIOUS_FAITH, RELIGIOUS_FAITH]}}
assert parse_packet(battery, json.dumps(dup)) is None
+ inconsistent = {"answers": valid["answers"],
+ "child_qualities": {"refused": True, "selected": [RELIGIOUS_FAITH]}}
+ assert parse_packet(battery, json.dumps(inconsistent)) is None
# refusal sentinel must not collide with a listed option
for q in battery:
assert REFUSED not in q["options"], q["id"]
@@ -607,7 +660,7 @@ def paid_smoke(battery: list[dict]) -> None:
raise SystemExit("paid smoke requires --i-authorize-paid-calls")
from wvs_score_all_options_refresh import reserve, settle_external_reservation
smoke_rid = f"pilot/resp-packet-smoke/{datetime.now(UTC).strftime('%Y%m%dT%H%M%S.%fZ')}"
- if not reserve({"id": smoke_rid, "lane": "google", "reserve_usd": "0.05"}):
+ if not reserve({"id": smoke_rid, "lane": "alibaba", "reserve_usd": "0.05"}):
raise RuntimeError("global repository cap rejected respondent-packet smoke reservation")
before = Decimal(update_state(lambda s: s)["conservative_spent_usd"])
smoke_records = OUT / "paid_smoke.jsonl"
@@ -648,6 +701,7 @@ def main() -> None:
actions.add_argument("--paid-smoke", action="store_true")
actions.add_argument("--run", action="store_true")
actions.add_argument("--analyze", action="store_true")
+ actions.add_argument("--synthetic-analysis-test", action="store_true")
parser.add_argument("--i-authorize-paid-calls", action="store_true",
help="required for --paid-smoke/--run; sets the paid-call opt-in")
args = parser.parse_args()
@@ -667,7 +721,9 @@ def main() -> None:
if args.run:
run(battery)
if args.analyze:
- analyze()
+ analyze_real()
+ if args.synthetic_analysis_test:
+ synthetic_analysis_test(battery, child_rows)
def run(battery: list[dict]) -> None:
@@ -687,8 +743,279 @@ def run(battery: list[dict]) -> None:
print("panel complete")
+def release_years() -> np.ndarray:
+ catalog = {r["id"]: r for r in json.loads(MODEL_CATALOG.read_text())["data"]}
+ def decimal_year(ts: int) -> float:
+ d = datetime.fromtimestamp(ts, tz=timezone.utc)
+ jan = datetime(d.year, 1, 1, tzinfo=timezone.utc)
+ nxt = datetime(d.year + 1, 1, 1, tzinfo=timezone.utc)
+ return d.year + (d - jan).total_seconds() / (nxt - jan).total_seconds()
+ return np.array([decimal_year(catalog[m]["created"]) for m in MODELS])
+
+
+class NoSubstantive(Exception):
+ """A bootstrap draw left an item with zero substantive rows."""
+
+
def analyze() -> None:
- raise SystemExit("analysis is implemented after the full run; see the preregistration")
+ raise SystemExit("replaced; see analyze_real")
+
+
+DENSE_LEDGER = Path("slop/research/wvs/20260916_openrouter/wvs_iw_requests.jsonl")
+
+
+def dense_qwen_psamples() -> dict[str, dict[str, list[np.ndarray]]]:
+ """model -> item suffix -> per-rating-sample p vectors, from the canonical dense-v1 ledger."""
+ run_ids = {e["model"]: e["run_id"] for e in
+ json.loads(DENSE_CACHE.read_text())["completed"].values()
+ if e.get("model") in MODELS}
+ if len(run_ids) != len(MODELS):
+ raise RuntimeError(f"dense cache lacks runs for: {set(MODELS) - set(run_ids)}")
+ out = {m: {} for m in MODELS}
+ for line in DENSE_LEDGER.read_text().splitlines():
+ r = json.loads(line)
+ if (r.get("event") == "item_result" and r.get("run_id") in run_ids.values()
+ and r.get("model") in MODELS):
+ out[r["model"]][r["id"]] = [np.asarray(v) for v in r["p_samples"]]
+ for m in MODELS:
+ if len(out[m]) != 12:
+ raise RuntimeError(f"dense item set incomplete for {m}: {len(out[m])}")
+ return out
+
+
+def packet_item_lists(rows: dict[int, dict], battery: list[dict],
+ child_rows: list[dict]) -> dict[str, list[np.ndarray]]:
+ """One vector per packet per scored item; refused packets contribute a zero vector (excluded
+ from the mean by the positivity of the mean, and coverage counts them)."""
+ from wvs_original_choice_pilot import child_binary_rows
+ binaries = child_binary_rows(child_rows)
+ out = {}
+ for q in battery:
+ if q["id"] == "ChildQualities":
+ for quality in PANEL_QUALITIES:
+ opts = binaries[quality]
+ vecs = []
+ for k in sorted(rows):
+ entry = rows[k]["ChildQualities"]
+ mentioned = entry["outcome"] == "substantive" and quality in entry["selected"]
+ vecs.append(np.array([1.0 if o == "Important" else 0.0 for o in opts])
+ if mentioned else
+ np.array([0.0 if o == "Important" else 1.0 for o in opts]))
+ out[quality] = vecs
+ continue
+ vecs = []
+ for k in sorted(rows):
+ entry = rows[k][q["id"]]
+ vec = np.zeros(len(q["options"]))
+ if entry["outcome"] == "substantive":
+ vec[q["options"].index(entry["selected"])] = 1.0
+ vecs.append(vec)
+ out[q["id"]] = vecs
+ return out
+
+
+def coords_draws(item_lists: dict[str, dict[str, list[np.ndarray]]], resolved: dict,
+ B: int = 1000, seed: int = 17, row_linked: bool = True) -> np.ndarray:
+ """(B, n_models, 2) coordinate draws.
+
+ row_linked=True (packet protocol): resample whole respondent rows, one index draw shared across
+ all items, preserving cross-question covariance. row_linked=False (dense v1): resample each
+ item's rating vectors independently; dense samples are not row-linked."""
+ n = len(next(iter(next(iter(item_lists.values())).values())))
+ rng = np.random.default_rng(seed)
+ out = np.empty((B, len(MODELS), 2))
+ for b in range(B):
+ shared = rng.integers(0, n, n)
+ for k, m in enumerate(MODELS):
+ xy = []
+ for axis in (X_AXIS, Y_AXIS):
+ vals = []
+ for it in resolved[axis]:
+ lists = item_lists[m][it["suffix"]]
+ idx = shared if row_linked else rng.integers(0, n, n)
+ mean_p = np.mean([lists[j] for j in idx], axis=0)
+ if mean_p.sum() == 0:
+ raise NoSubstantive(f"{m} {it['suffix']} has no substantive rows in this draw")
+ vals.append(positiveness(mean_p[None, :], it["pole_idx"], it["n"]))
+ xy.append(float(np.mean(vals)))
+ out[b, k] = xy
+ return out
+
+
+def constant_and_linear_rmse(coords: np.ndarray) -> tuple[float, float]:
+ """2D residual RMSE of the n_models coordinates around (a) the family centroid and (b) the
+ release-date OLS line."""
+ x = release_years()
+ centroid = coords.mean(axis=0)
+ constant = float(np.sqrt(np.mean(np.sum((coords - centroid) ** 2, axis=1))))
+ residuals = []
+ for k in range(2):
+ slope, intercept = np.polyfit(x, coords[:, k], 1)
+ residuals.append(coords[:, k] - (slope * x + intercept))
+ linear = float(np.sqrt(np.mean(np.sum(np.stack(residuals, axis=1) ** 2, axis=1))))
+ return constant, linear
+
+
+def loo_prediction_error(coords: np.ndarray) -> dict:
+ """Leave-one-release-out prediction error (mean 2D distance) for the constant (family centroid
+ of the other releases) vs linear (OLS on the other releases) predictor. With n=4 the linear
+ fit has 1 residual dof and is expected to overfit; this makes that visible."""
+ x = release_years()
+ const_errs, lin_errs = [], []
+ for i in range(len(MODELS)):
+ others = [j for j in range(len(MODELS)) if j != i]
+ const_pred = coords[others].mean(axis=0)
+ slopes = [np.polyfit(x[others], coords[others, k], 1) for k in range(2)]
+ lin_pred = np.array([slopes[k][0] * x[i] + slopes[k][1] for k in range(2)])
+ const_errs.append(float(np.linalg.norm(coords[i] - const_pred)))
+ lin_errs.append(float(np.linalg.norm(coords[i] - lin_pred)))
+ return {"constant_loo_mean": float(np.mean(const_errs)),
+ "linear_loo_mean": float(np.mean(lin_errs)),
+ "constant_loo_per_release": const_errs, "linear_loo_per_release": lin_errs}
+
+
+def analyze_from_data(packet_rows: dict[str, dict[int, dict]], battery: list[dict],
+ child_rows: list[dict], resolved: dict) -> dict:
+ """The fixed preregistered analysis, run on real or synthetic rows."""
+ per_model, item_lists = {}, {}
+ for m in MODELS:
+ rows = packet_rows[m]
+ if len(rows) != N_PACKETS:
+ raise RuntimeError(f"{m}: expected {N_PACKETS} rows, got {len(rows)}")
+ il = packet_item_lists(rows, battery, child_rows)
+ item_lists[m] = il
+ refused = {q["id"]: sum(1 for k in sorted(rows) if rows[k][q["id"]]["outcome"] == "refused")
+ for q in battery}
+ per_model[m] = {"refused_per_question": refused,
+ "refusal_rate": sum(refused.values()) / (len(refused) * N_PACKETS),
+ "coverage_per_question": {q["id"]: 1 - refused[q["id"]] / N_PACKETS
+ for q in battery}}
+ draws = coords_draws(item_lists, resolved)
+ point = draws.mean(axis=0)
+ se = draws.std(axis=0)
+ constant, linear = constant_and_linear_rmse(point)
+ # bootstrap scatter distributions
+ constant_draws = np.empty(draws.shape[0]); linear_draws = np.empty(draws.shape[0])
+ for b in range(draws.shape[0]):
+ constant_draws[b], linear_draws[b] = constant_and_linear_rmse(draws[b])
+ # noise floors: H0 = no between-release differences; draw each release from N(0, SE)
+ rng = np.random.default_rng(23)
+ floor_const, floor_lin = np.empty(2000), np.empty(2000)
+ for b in range(2000):
+ ys = np.array([rng.normal(0.0, se[k]) for k in range(len(MODELS))])
+ floor_const[b], floor_lin[b] = constant_and_linear_rmse(ys)
+ dense = dense_qwen_psamples()
+ dense_item_lists = {m: {k: list(v) for k, v in dense[m].items()} for m in MODELS}
+ dense_draws = coords_draws(dense_item_lists, resolved, seed=29, row_linked=False)
+ dense_point = dense_draws.mean(axis=0)
+ dense_se = dense_draws.std(axis=0)
+ dense_constant, dense_linear = constant_and_linear_rmse(dense_point)
+ dense_constant_draws = np.empty(draws.shape[0]); dense_linear_draws = np.empty(draws.shape[0])
+ for b in range(draws.shape[0]):
+ dense_constant_draws[b], dense_linear_draws[b] = constant_and_linear_rmse(dense_draws[b])
+ rng2 = np.random.default_rng(31)
+ dense_floor_const, dense_floor_lin = np.empty(2000), np.empty(2000)
+ for b in range(2000):
+ ys = np.array([rng2.normal(0.0, dense_se[k]) for k in range(len(MODELS))])
+ dense_floor_const[b], dense_floor_lin[b] = constant_and_linear_rmse(ys)
+ release_years_list = release_years().tolist()
+ return {
+ "eval_version": EVAL_VERSION, "n_packets": N_PACKETS,
+ "release_years": release_years_list,
+ "packet": {
+ "coords_xy_per_model": {m: [float(v) for v in point[k]] for k, m in enumerate(MODELS)},
+ "coord_se": {m: [float(v) for v in se[k]] for k, m in enumerate(MODELS)},
+ "constant_family_rmse": constant, "constant_family_rmse_se": float(np.std(constant_draws)),
+ "linear_trend_rmse": linear, "linear_trend_rmse_se": float(np.std(linear_draws)),
+ "constant_noise_floor_mean": float(np.mean(floor_const)),
+ "linear_noise_floor_mean": float(np.mean(floor_lin)),
+ "p_constant_noise_ge_observed": float(np.mean(floor_const >= constant)),
+ "p_linear_noise_ge_observed": float(np.mean(floor_lin >= linear)),
+ "loo": loo_prediction_error(point),
+ "refusal": {m: per_model[m]["refusal_rate"] for m in MODELS},
+ "coverage": {m: per_model[m]["coverage_per_question"] for m in MODELS},
+ },
+ "dense_v1": {
+ "coords_xy_per_model": {m: [float(v) for v in dense_point[k]] for k, m in enumerate(MODELS)},
+ "coord_se": {m: [float(v) for v in dense_se[k]] for k, m in enumerate(MODELS)},
+ "constant_family_rmse": dense_constant,
+ "constant_family_rmse_se": float(np.std(dense_constant_draws)),
+ "linear_trend_rmse": dense_linear, "linear_trend_rmse_se": float(np.std(dense_linear_draws)),
+ "constant_noise_floor_mean": float(np.mean(dense_floor_const)),
+ "linear_noise_floor_mean": float(np.mean(dense_floor_lin)),
+ "p_constant_noise_ge_observed": float(np.mean(dense_floor_const >= dense_constant)),
+ "p_linear_noise_ge_observed": float(np.mean(dense_floor_lin >= dense_linear)),
+ "loo": loo_prediction_error(dense_point),
+ "bootstrap_note": ("item-wise independent resampling of the 6 rating vectors per item; "
+ "dense samples are not row-linked, so protocols are not pairable and "
+ "differences use independent bootstraps"),
+ },
+ "coord_shifts_packet_minus_dense": {m: [float(a - b) for a, b in zip(point[k], dense_point[k])]
+ for k, m in enumerate(MODELS)},
+ }
+
+
+def load_wvs_recs() -> list[dict]:
+ from wvs_map import load_wvs_all
+ return load_wvs_all()
+
+
+def analyze_real() -> None:
+ battery, child_rows = build_battery()
+ resolved = resolve_items(load_wvs_recs())
+ packet_rows = {}
+ for m in MODELS:
+ rpath = OUT / "records" / m.replace("/", "__") / "packets.jsonl"
+ packet_rows[m] = prior_rows(rpath, protocol_id(m, battery))
+ analysis = analyze_from_data(packet_rows, battery, child_rows, resolved)
+ atomic_json(OUT / "analysis.json", analysis)
+ print(json.dumps({"packet": {k: analysis["packet"][k]
+ for k in ("constant_family_rmse", "linear_trend_rmse", "loo")},
+ "dense": {k: analysis["dense_v1"][k]
+ for k in ("constant_family_rmse", "linear_trend_rmse", "loo")}},
+ indent=2))
+
+
+def synthetic_analysis_test(battery: list[dict], child_rows: list[dict]) -> None:
+ """End-to-end analysis on synthetic data: refusal semantics, covariance preservation, fits,
+ floors, and LOO all exercise without any paid call."""
+ resolved = resolve_items(load_wvs_recs())
+ rng = np.random.default_rng(5)
+ options = {q["id"]: q["options"] for q in battery}
+ packet_rows = {}
+ for k, m in enumerate(MODELS):
+ rows = {}
+ for p in range(N_PACKETS):
+ row = {}
+ for q in battery:
+ if q["id"] == "ChildQualities":
+ pool = q["options"]
+ if rng.random() < 0.03:
+ row[q["id"]] = {"outcome": "refused", "selected": []}
+ else:
+ picked = list(rng.choice(pool, size=rng.integers(0, 6), replace=False))
+ row[q["id"]] = {"outcome": "substantive", "selected": picked}
+ else:
+ if rng.random() < 0.05 * (k + 1):
+ row[q["id"]] = {"outcome": "refused", "selected": REFUSED}
+ else:
+ # drift with release index k so the linear fit has signal
+ j = min(len(q["options"]) - 1, max(0, int(rng.normal(k * 0.8, 1.5))))
+ row[q["id"]] = {"outcome": "substantive", "selected": q["options"][j]}
+ rows[p] = row
+ packet_rows[m] = rows
+ result = analyze_from_data(packet_rows, battery, child_rows, resolved)
+ for section in ("packet", "dense_v1"):
+ for key in ("constant_family_rmse", "linear_trend_rmse", "loo"):
+ value = result[section][key]
+ assert np.isfinite(value if not isinstance(value, dict) else
+ value["constant_loo_mean"] + value["linear_loo_mean"]), (section, key)
+ assert result["packet"]["refusal"][MODELS[3]] > result["packet"]["refusal"][MODELS[0]]
+ print(f"synthetic analysis passed: packet constant={result['packet']['constant_family_rmse']:.4f} "
+ f"linear={result['packet']['linear_trend_rmse']:.4f}; "
+ f"dense constant={result['dense_v1']['constant_family_rmse']:.4f}; "
+ f"packet LOO const={result['packet']['loo']['constant_loo_mean']:.4f} "
+ f"lin={result['packet']['loo']['linear_loo_mean']:.4f}")
if __name__ == "__main__":
diff --git a/slop/research/wvs/20260919_respondent_packet/endpoint_catalog_live.json b/slop/research/wvs/20260919_respondent_packet/endpoint_catalog_live.json
index 2fbd669..12a92a3 100644
--- a/slop/research/wvs/20260919_respondent_packet/endpoint_catalog_live.json
+++ b/slop/research/wvs/20260919_respondent_packet/endpoint_catalog_live.json
@@ -1,107 +1,5 @@
{
- "fetched_utc": "2026-09-19T05:03:39.413618+00:00",
- "models": {
- "qwen/qwen3.5-plus-02-15": {
- "alibaba_endpoints": [],
- "created": 1771229416,
- "reasoning": {
- "mandatory": false
- },
- "supported_parameters": [
- "frequency_penalty",
- "include_reasoning",
- "logprobs",
- "max_tokens",
- "presence_penalty",
- "reasoning",
- "response_format",
- "seed",
- "stop",
- "structured_outputs",
- "temperature",
- "tool_choice",
- "tools",
- "top_k",
- "top_logprobs",
- "top_p"
- ]
- },
- "qwen/qwen3.5-plus-20260420": {
- "alibaba_endpoints": [],
- "created": 1777261368,
- "reasoning": {
- "mandatory": false
- },
- "supported_parameters": [
- "frequency_penalty",
- "include_reasoning",
- "logprobs",
- "max_tokens",
- "presence_penalty",
- "reasoning",
- "response_format",
- "seed",
- "stop",
- "structured_outputs",
- "temperature",
- "tool_choice",
- "tools",
- "top_k",
- "top_logprobs",
- "top_p"
- ]
- },
- "qwen/qwen3.6-plus": {
- "alibaba_endpoints": [],
- "created": 1775133557,
- "reasoning": {
- "mandatory": false
- },
- "supported_parameters": [
- "frequency_penalty",
- "include_reasoning",
- "logprobs",
- "max_tokens",
- "presence_penalty",
- "reasoning",
- "response_format",
- "seed",
- "stop",
- "structured_outputs",
- "temperature",
- "tool_choice",
- "tools",
- "top_k",
- "top_logprobs",
- "top_p"
- ]
- },
- "qwen/qwen3.7-plus": {
- "alibaba_endpoints": [],
- "created": 1780491783,
- "reasoning": {
- "default_enabled": true,
- "mandatory": false
- },
- "supported_parameters": [
- "frequency_penalty",
- "include_reasoning",
- "logprobs",
- "max_tokens",
- "presence_penalty",
- "reasoning",
- "response_format",
- "seed",
- "stop",
- "structured_outputs",
- "temperature",
- "tool_choice",
- "tools",
- "top_k",
- "top_logprobs",
- "top_p"
- ]
- }
- },
- "schema": 1
+ "diagnosis": "An unauthenticated GET https://openrouter.ai/api/v1/models on 2026-09-19 returned endpoints=null for every model, so this file supplies NO endpoint evidence. Endpoints and pricing come only from the authenticated snapshot slop/research/wvs/20260918_gemini_flash_rubric_pilot/endpoint_catalog.json (fetched_utc 2026-09-18T10:50:27Z, one Alibaba endpoint per Qwen Plus model, advertised quantization unknown). The paid smoke is the current live route test; each response's openrouter_metadata is the live route evidence.",
+ "fetched_utc_of_null_endpoint_response": "2026-09-19T05:03Z",
+ "supplies_endpoint_evidence": false
}
diff --git a/slop/research/wvs/20260919_respondent_packet/manifest.json b/slop/research/wvs/20260919_respondent_packet/manifest.json
index 2e1d481..bae518b 100644
--- a/slop/research/wvs/20260919_respondent_packet/manifest.json
+++ b/slop/research/wvs/20260919_respondent_packet/manifest.json
@@ -103,7 +103,8 @@
"question": "Here is a list of qualities that children can be encouraged to learn at home. Which, if any, do you consider to be especially important? Please choose up to five."
}
],
- "created_utc": "2026-09-19T05:08:22.897161+00:00",
+ "created_utc": "2026-09-19T05:23:13.067458+00:00",
+ "endpoint_provenance": "endpoints and pricing from the authenticated 2026-09-18 snapshot; the unauthenticated live models GET returns endpoints=null, so the paid smoke is the current live route test",
"eval_version": "wvs-respondent-packet-v1",
"instrument_status": "GlobalOpinionQA-compatible approximation; 11th quality Religious faith appended; card order not recorded in source",
"max_tokens": 2048,
@@ -119,7 +120,7 @@
},
"release_slug": "qwen/qwen3.5-plus-20260216",
"requests": 128,
- "reserve_bound_usd": "0.88604672"
+ "reserve_bound_usd": "0.44302336"
},
{
"advertised_quantization": "unknown",
@@ -131,7 +132,7 @@
},
"release_slug": "qwen/qwen3.6-plus-04-02",
"requests": 128,
- "reserve_bound_usd": "1.107558400"
+ "reserve_bound_usd": "0.553779200"
},
{
"advertised_quantization": "unknown",
@@ -143,7 +144,7 @@
},
"release_slug": "qwen/qwen3.5-plus-20260420",
"requests": 128,
- "reserve_bound_usd": "1.0223616"
+ "reserve_bound_usd": "0.5111808"
},
{
"advertised_quantization": "unknown",
@@ -155,18 +156,20 @@
},
"release_slug": "qwen/qwen3.7-plus-20260602",
"requests": 128,
- "reserve_bound_usd": "0.75497472"
+ "reserve_bound_usd": "0.37748736"
}
],
"n_packets": 128,
"not_published": true,
- "panel_plus_smoke_bound_usd": "3.777863680",
+ "panel_plus_smoke_bound_usd": "1.888419840",
"panel_requests": 512,
- "panel_reserve_bound_usd": "3.770941440",
- "refusal_rule": "per-question \"refused\" status inside the schema; never displayed as an option; reported separately; never neutral",
- "schema": 1,
+ "panel_reserve_bound_usd": "1.885470720",
+ "refusal_rule": "ordinary: per-question \"refused\" status inside the schema, never displayed as an option; child: separate refused flag plus a 0..5 selected list, zero selections with refused=false is a valid substantive none; all refusals reported separately, never neutral",
+ "reserve_bound_formula": "1024 input tokens + 2048 output tokens per packet at the saved Alibaba per-token prices",
+ "schema": 2,
+ "smoke_model": "qwen/qwen3.7-plus",
"smoke_requests": 1,
- "smoke_reserve_bound_usd": "0.00692224",
+ "smoke_reserve_bound_usd": "0.00294912",
"stage_hard_stop_usd": "5",
"temperature": 1.0
}