diff --git a/docs/RESEARCH_JOURNAL.md b/docs/RESEARCH_JOURNAL.md index 4170b51..cab2726 100644 --- a/docs/RESEARCH_JOURNAL.md +++ b/docs/RESEARCH_JOURNAL.md @@ -1148,7 +1148,7 @@ Pueue task 1706 completed all five Gemini Flash releases on the preregistered or Evidence: `cannot_answer` rates are 76/216 (35.2%) Preview, 16/216 (7.4%) 3.5, 60/216 (27.8%) 3.6, 118/216 (54.6%) 3.7, 87/216 (40.3%) 3.8. Abortion drew `cannot_answer` from 100% of samples in four of five releases; God from 100% in 3.7 and 3.8. Conditional coordinates with paired-bootstrap SE (B=1000): Preview (0.9125, 0.6806), 3.5 (0.9083, 0.5602), 3.6 (0.6042, 0.6273), 3.7 (0.5542, 0.9000), 3.8 (0.4875, 0.9083); SEs 0.020-0.058. Release-date OLS: x slope -0.6062/year (SE 0.0496), y slope +0.2834/year (SE 0.0486), 2D residual RMSE 0.1541 (SE 0.0122). -Interpretation, separated: against the preregistered questions, the single-choice original format does not reduce abstention (rates overlap the dense 35-48% flat range and are more heterogeneous) and increases release-date scatter (RMSE 0.1541 versus 0.0518-0.0639 dense, about 2.4-3.0x). Doubt migrates into the explicit non-substantive options instead of disappearing; the preregistered P1 (nonzero, heterogeneous explicit refusal) is supported, P2 (original differs from dense min) is supported on most releases, and P3 (no directional prediction) observed a worse direction for scatter. Limitations carry over: release order, model identity, and fixed wall-clock order are exactly confounded; n=5; 3.7/3.8 y rests on 5 of 7 Y items due to zero God/Abortion coverage. Per the preregistration, no protocol is selected on trend. -- PI[gpt-5.6-terra] +Interpretation, separated: against the preregistered questions, the single-choice original format does not clearly reduce total abstention (explicit refusal rates 7.4-54.6% sit inside the dense 35-48% flat range, but dense flat vectors are evidence of POSSIBLE hidden abstention, not observed refusals, so the two rates are not equivalent quantities; what the original format does is make part of the abstention explicit and measurable), and it increases release-date scatter: on fixed common item sets (the corrected estimand, dense comparators recomputed on exactly the same items) the original-choice 2D RMSE is 0.1232 vs dense min 0.0627 / high 0.0521 (common nonzero-coverage rule, 5 X + 5 Y items), 0.1691 vs 0.1106/0.0937 (>= 25% coverage), and 0.2643 vs 0.1259/0.1745 (>= 50%), i.e. about 2.0-2.4x the dense minimum everywhere a comparison exists. Preregistered P1 (nonzero, heterogeneous explicit refusal) and P2 (original differs from dense min) are supported; P3 observed a worse direction for scatter. The all-items five-release fit is exploratory only: the gemini-3.8-flash panel ran 11 minutes before the preregistration commit, so it is labeled exploratory/unpreregistered and the five-point fit is not a preregistered prediction test. Option-position/refusal contrasts show near-uniform selection across position thirds and near-flat refusal rates within release across DK positions, ruling out cyclic-order artifacts for the release-level patterns. Limitations carry over: release order, model identity, and fixed wall-clock order are exactly confounded; n=5; the cov50 X axis is a single item. Per the preregistration, no protocol is selected on trend. A fail-fast paid-call opt-in now guards the runner, with an offline regression proving a mis-targeted monkeypatch cannot reach the API again; root cause was a `from ... import` binding that a read_api-level patch never rebinds. -- PI[gpt-5.6-terra] Source: `slop/research/wvs/20260918_original_choice_pilot/` (`results.json`, `analysis.json`, raw `records/`, `request_attempts.jsonl`, `budget.json`, `pueue_task_1706_clean.log`) and the global ledger `slop/research/wvs/20260917_score_all_options/budget.json`. Nothing published; the map is unchanged. ## 2026-09-18 -- Preregistration: wvs-original-choice-pilot-v1 diff --git a/scripts/wvs_api/09_original_choice_pilot.sh b/scripts/wvs_api/09_original_choice_pilot.sh index 1a1a8bb..e65c8c3 100755 --- a/scripts/wvs_api/09_original_choice_pilot.sh +++ b/scripts/wvs_api/09_original_choice_pilot.sh @@ -1,3 +1,3 @@ #!/bin/sh set -eu -exec uv run --offline --with 'datasets>=4.0,<5' python scripts/wvs_original_choice_pilot.py --run +exec uv run --offline --with 'datasets>=4.0,<5' python scripts/wvs_original_choice_pilot.py --run --i-authorize-paid-calls diff --git a/scripts/wvs_original_choice_pilot.py b/scripts/wvs_original_choice_pilot.py index 8614c52..b84a64a 100644 --- a/scripts/wvs_original_choice_pilot.py +++ b/scripts/wvs_original_choice_pilot.py @@ -48,6 +48,21 @@ N_SAMPLES = 24 MAX_TOKENS = 1024 REQUEST_TIMEOUT = 400 STAGE_CAP_USD = Decimal("5") +PAID_OPTIN_ENV = "WVS_PAID_CALLS_AUTHORIZED" + + +class PaidCallsNotAuthorized(RuntimeError): + """Raised when a paid request is attempted without the explicit opt-in. + + Root cause of the 2026-09-19 unplanned gemini-3.8 panel: the test monkeypatched + `moralmaps.read_api.openrouter_request_with_metadata_once`, but this module had already bound + its own name via `from ... import` at import time, so the patch never took effect and the real + API was called. The opt-in guard is a second, independent line of defense: even a mis-patched + or unpatched request path refuses to execute unless the run was explicitly authorized.""" + + +def paid_calls_authorized() -> bool: + return os.environ.get(PAID_OPTIN_ENV) == "1" OUT = Path("slop/research/wvs/20260918_original_choice_pilot") MANIFEST = OUT / "manifest.json" RESULTS = OUT / "results.json" @@ -275,6 +290,10 @@ def append_attempt(record: dict) -> None: async def budgeted_request(model: str, payload: dict) -> dict: + if not paid_calls_authorized(): + raise PaidCallsNotAuthorized( + f"paid calls require --i-authorize-paid-calls (which sets {PAID_OPTIN_ENV}=1); " + "offline tests must never reach the API") payload_hash = hashlib.sha256(json.dumps(payload, sort_keys=True).encode()).hexdigest() for attempt in range(1, MAX_ATTEMPTS + 1): bound = reserve_request(model) @@ -625,6 +644,57 @@ def write_manifest(items: list[dict]) -> dict: return payload +def offline_regression(items: list[dict]) -> None: + """Proves the paid path cannot execute without the opt-in, even with the real request function + in place and a transport that would succeed. Regression for the 2026-09-19 accident.""" + import moralmaps.read_api as ra + os.environ.pop(PAID_OPTIN_ENV, None) + assert not paid_calls_authorized() + calls = {"n": 0} + + def would_reach_api(request: httpx.Request) -> httpx.Response: + calls["n"] += 1 + return httpx.Response(200, json={"choices": [], "usage": {}}) + + payload = {"model": MODELS[0], "messages": [{"role": "user", "content": "test"}]} + original = ra.openrouter_request_with_metadata_once + old_api_key = os.environ.pop("OPENROUTER_API_KEY", None) + os.environ["OPENROUTER_API_KEY"] = "test-key" + try: + ran = False + try: + asyncio.run(budgeted_request(MODELS[0], payload)) + except PaidCallsNotAuthorized: + ran = True + assert ran, "budgeted_request executed without the opt-in" + assert calls["n"] == 0, "a paid HTTP call leaked without the opt-in" + # even the raw request function, when invoked through the real transport path, is only + # reachable via budgeted_request inside this module; a mis-targeted read_api patch cannot + # bypass the guard: + async def patched(payload, timeout=60.0, **kwargs): + return await original(payload, timeout=timeout, transport=httpx.MockTransport(would_reach_api)) + ra.openrouter_request_with_metadata_once = patched + ran = False + try: + asyncio.run(budgeted_request(MODELS[0], payload)) + except PaidCallsNotAuthorized: + ran = True + assert ran and calls["n"] == 0, "opt-in guard bypassed by a read_api-level patch" + finally: + ra.openrouter_request_with_metadata_once = original + if old_api_key is not None: + os.environ["OPENROUTER_API_KEY"] = old_api_key + else: + os.environ.pop("OPENROUTER_API_KEY", None) + # CLI guard: --run without the flag must fail fast + import subprocess + proc = subprocess.run(["uv", "run", "--offline", "--with", "datasets>=4.0,<5", "python", + "scripts/wvs_original_choice_pilot.py", "--run"], + capture_output=True, text=True, timeout=300) + assert proc.returncode != 0 and "--i-authorize-paid-calls" in (proc.stderr + proc.stdout) + print("offline regression passed: no paid call without opt-in, mis-patch cannot bypass, CLI guard holds") + + def offline_smoke(items: list[dict]) -> None: ordinary = next(i for i in items if not i["is_list"]) order = presented(items, 0, 1) @@ -737,6 +807,131 @@ def load_wvs_recs() -> list[dict]: return load_wvs_all() +DENSE_RESULTS = Path("slop/research/wvs/20260918_gemini_flash_rubric_pilot/results.json") + + +def dense_psamples(cell_name: str) -> dict[str, dict[str, np.ndarray]]: + """model -> item suffix -> per-sample p over the canonical substantive options, from the dense + rubric pilot (canonical score-all-options, normal rubric).""" + data = json.loads(DENSE_RESULTS.read_text()) + out = {} + for row in data["models"]: + cell = next(c for c in row["cells"] if c["name"] == cell_name) + events = [json.loads(line) for line in Path(cell["records"]).read_text().splitlines()] + finished = [e for e in events if e["event"] == "run_finished"][-1] + out[row["id"]] = {e["id"]: np.asarray(e["p_samples"]) for e in events + if e["event"] == "item_result" and e["run_id"] == finished["run_id"]} + return out + + +def release_years() -> np.ndarray: + catalog = {r["id"]: r for r in json.loads(MODEL_CATALOG.read_text())["data"]} + def decimal_year(ts: int) -> float: + d = datetime.fromtimestamp(ts, tz=timezone.utc) + jan = datetime(d.year, 1, 1, tzinfo=timezone.utc) + nxt = datetime(d.year + 1, 1, 1, tzinfo=timezone.utc) + return d.year + (d - jan).total_seconds() / (nxt - jan).total_seconds() + return np.array([decimal_year(catalog[m]["created"]) for m in MODELS]) + + +def coords_on_items(per_item: dict[str, dict[str, np.ndarray]], resolved: dict, + item_ids: set[str]) -> dict: + """Coordinates, release OLS slopes, and 2D residual RMSE restricted to a fixed item set.""" + x = release_years() + ys = [] + for m in MODELS: + xy = [] + for axis in (X_AXIS, Y_AXIS): + vals = [positiveness(np.mean(per_item[m][it["suffix"]], axis=0)[None, :], + it["pole_idx"], it["n"]) + for it in resolved[axis] if it["suffix"] in item_ids] + xy.append(float(np.mean(vals))) + ys.append(xy) + ys = np.array(ys) + slopes, preds = [], [] + for k in range(2): + slope, intercept = np.polyfit(x, ys[:, k], 1) + slopes.append(float(slope)) + preds.append(slope * x + intercept) + rmse2 = float(np.sqrt(np.mean(np.sum((ys - np.column_stack(preds)) ** 2, axis=1)))) + return {"coords_xy_per_model": {m: [float(v) for v in ys[k]] for k, m in enumerate(MODELS)}, + "slope_x_per_year": slopes[0], "slope_y_per_year": slopes[1], "rmse_2d": rmse2} + + +def sensitivity(items: list[dict], child_rows: list[dict], resolved: dict, + summaries: dict[str, dict]) -> dict: + """Release fits on fixed common item sets (coverage rule applied to ALL five original-choice + releases simultaneously), with dense normal_minimum/high recomputed on exactly the same items. + This fixes the estimand the all-items comparison changed.""" + plan = plan_requests(items) + orig = {m: sample_psamples(m, items, Path(summaries[m]["records"]), summaries[m]["protocol_id"], + plan, child_rows) for m in MODELS} + dense_min = dense_psamples("normal_minimum") + dense_high = dense_psamples("normal_high") + all_suffixes = sorted(orig[MODELS[0]]) + rules = {"nonzero": 0.0, "cov25": 0.25, "cov50": 0.50, "cov75": 0.75} + out = {} + for rule, threshold in rules.items(): + keep = [s for s in all_suffixes + if all(sum(v.sum() > 0 for v in orig[m][s]) / len(orig[m][s]) > threshold + for m in MODELS)] + keep = set(keep) + counts = {axis: sorted(it["suffix"] for it in resolved[axis] if it["suffix"] in keep) + for axis in (X_AXIS, Y_AXIS)} + out[rule] = {"min_per_model_coverage": threshold, + "items": counts, + "n_items": {axis: len(counts[axis]) for axis in (X_AXIS, Y_AXIS)}, + "original_choice": coords_on_items(orig, resolved, keep), + "dense_normal_minimum": coords_on_items(dense_min, resolved, keep), + "dense_normal_high": coords_on_items(dense_high, resolved, keep)} + return out + + +def position_contrast(items: list[dict]) -> dict: + """Balance evidence and option-position/refusal contrast from the records: rotations cover each + position equally; cannot_answer rate by the position of the nearest non-substantive option, and + substantive selection rate by normalized position bucket (first/middle/last third).""" + out = {} + for model in MODELS: + rpath = OUT / "records" / model.replace("/", "__") / "original_choice.jsonl" + answers = [e for e in load_events(rpath) if e["event"] == "answer_parsed"] + rows = [] + for q, item in enumerate(items): + if item["is_list"]: + continue + k = len(item["offered"]) + base = item["offered"] + for e in answers: + if e["item_id"] != item["id"]: + continue + s = e["sample"] + pos_of = {opt: (opt_pos - s) % k for opt_pos, opt in enumerate(base)} + dk_rank = min(pos_of[o] for o in NONSUBSTANTIVE) / k + if e["outcome"] == "substantive": + sel_rank = pos_of[e["selected"]] / k + rows.append((dk_rank, sel_rank, False)) + else: + rows.append((dk_rank, None, True)) + def bucket(rank): + return "first_third" if rank < 1 / 3 else ("middle" if rank < 2 / 3 else "last_third") + cannot_by_dk_position = {b: [0, 0] for b in ("first_third", "middle", "last_third")} + sel_by_position = {b: [0, 0] for b in ("first_third", "middle", "last_third")} + for dk_rank, sel_rank, cannot in rows: + cannot_by_dk_position[bucket(dk_rank)][1] += 1 + cannot_by_dk_position[bucket(dk_rank)][0] += int(cannot) + if sel_rank is not None: + sel_by_position[bucket(sel_rank)][1] += 1 + out[model] = { + "cannot_answer_rate_by_dk_position_third": { + b: (round(v[0] / v[1], 4) if v[1] else None) for b, v in cannot_by_dk_position.items()}, + "n_by_dk_position_third": {b: v[1] for b, v in cannot_by_dk_position.items()}, + "substantive_selection_share_by_position_third": { + b: round(v[1] / sum(x[1] for x in sel_by_position.values()), 4) + for b, v in sel_by_position.items()}, + } + return out + + def analyze() -> None: """Post-run analysis: coordinates, coverage, refusal rates, paired bootstrap, release trend.""" items, child_rows = build_items() @@ -751,27 +946,52 @@ def analyze() -> None: coverage = {m: {item_id: stats["coverage"] for item_id, stats in summaries[m]["per_question"].items()} for m in MODELS} atomic_json(OUT / "analysis.json", { - "fits": fits, + "schema": 2, + "status_labels": { + "gemini-3-flash-preview": "preregistered", "google/gemini-3.5-flash": "preregistered", + "google/gemini-3.6-flash": "preregistered", "google/gemini-3.7-flash": "smoke-then-preregistered-panel", + "google/gemini-3.8-flash": "EXPLORATORY, ran before the preregistration commit 5ad68ad"}, + "note": ("the all-items five-release fit below is therefore NOT a preregistered prediction " + "test; preregistered claims are per-release P1/P2 only"), + "fits_all_items_exploratory": fits, + "sensitivity_fixed_item_sets": sensitivity(items, child_rows, resolved, summaries), + "position_contrast": position_contrast(items), "cannot_answer_rate_per_model": {m: sum(cannot[m].values()) / len(cannot[m]) for m in MODELS}, "cannot_answer_rate_per_question_per_model": cannot, "coverage_per_question_per_model": coverage, }) - print(json.dumps(fits["point"], indent=2)) + print(json.dumps({"all_items": fits["point"], + "sensitivity": {k: {"n_items": v["n_items"], + "orig_rmse": v["original_choice"]["rmse_2d"], + "dense_min_rmse": v["dense_normal_minimum"]["rmse_2d"], + "dense_high_rmse": v["dense_normal_high"]["rmse_2d"]} + for k, v in sensitivity(items, child_rows, resolved, summaries).items()}}, + indent=2)) def main() -> None: parser = argparse.ArgumentParser() - parser.add_argument("--offline-smoke", action="store_true") - parser.add_argument("--write-manifest", action="store_true") - parser.add_argument("--paid-smoke", action="store_true") - parser.add_argument("--run", action="store_true") - parser.add_argument("--analyze", action="store_true") + actions = parser.add_mutually_exclusive_group(required=True) + actions.add_argument("--offline-smoke", action="store_true") + actions.add_argument("--offline-regression", action="store_true") + actions.add_argument("--write-manifest", action="store_true") + actions.add_argument("--paid-smoke", action="store_true") + actions.add_argument("--run", action="store_true") + actions.add_argument("--analyze", action="store_true") + parser.add_argument("--i-authorize-paid-calls", action="store_true", + help="required for --paid-smoke/--run; sets the paid-call opt-in") args = parser.parse_args() items, child_rows = build_items() if args.offline_smoke: offline_smoke(items) + if args.offline_regression: + offline_regression(items) if args.write_manifest: write_manifest(items) + if args.paid_smoke or args.run: + if not args.i_authorize_paid_calls: + raise SystemExit("--paid-smoke/--run cost money and require --i-authorize-paid-calls") + os.environ[PAID_OPTIN_ENV] = "1" if args.paid_smoke: paid_smoke(items) if args.run: diff --git a/slop/audits/job_1706_original_choice_pilot.md b/slop/audits/job_1706_original_choice_pilot.md index 448e193..8fd5790 100644 --- a/slop/audits/job_1706_original_choice_pilot.md +++ b/slop/audits/job_1706_original_choice_pilot.md @@ -1,5 +1,10 @@ # Audit: Pueue job 1706, original-choice pilot (wvs-original-choice-pilot-v1) +## Status labels (correction 1) + +The gemini-3.8-flash panel started 2026-09-19 07:20 +0800, while preregistration commit `5ad68ad40c4cf2d223e6d7554dd4e2b520dfb9ad` landed 07:31 +0800. The 3.8 panel is therefore **exploratory / unpreregistered**; it is retained as valid protocol output but the all-items five-release fit is NOT a preregistered prediction test. The preregistered claims are the per-release P1/P2 directions on the four later releases (3.7's smoke ran at 23:30 UTC on 09-18, after the prereg; its full panel and the other three panels ran after `5ad68ad`). + + ## Target and provenance Pueue task 1706 ran `scripts/wvs_api/09_original_choice_pilot.sh` from 2026-09-19 07:31:23 to 07:56:26 +0800 and exited `Success` in 1503 s. The runner is `scripts/wvs_original_choice_pilot.py` (EVAL_VERSION `wvs-original-choice-pilot-v1`), preregistered in `docs/RESEARCH_JOURNAL.md` ("Preregistration: wvs-original-choice-pilot-v1") and committed at `5ad68ad40c4cf2d223e6d7554dd4e2b520dfb9ad` before the queued paid calls. Branch `research/gemini-flash-rubric-v1`; published map untouched. @@ -28,24 +33,49 @@ Hard stage stop was USD 5; headroom USD 4.69. Global reservations are empty. `cannot_answer` is explicit and heterogeneous (overall rate per release): Preview 76/216 (35.2%), 3.5 16/216 (7.4%), 3.6 60/216 (27.8%), 3.7 118/216 (54.6%), 3.8 87/216 (40.3%). Zero-coverage items: Abortion for Preview and 3.6; God and Abortion for 3.7 and 3.8. Abortion is `cannot_answer` for 100% of samples in four of five releases. -Conditional coordinates (x = Survival<->Self-expression, y = Traditional<->Secular-Rational), with paired-bootstrap SE over the 24 shared sample indices (B=1000, all draws used): +Conditional coordinates (x = Survival<->Self-expression, y = Traditional<->Secular-Rational), with paired-bootstrap SE over the 24 shared sample indices (B=1000, all draws used). The 3.8 row is exploratory: -| release | x (SE) | y (SE) | -| --- | ---: | ---: | -| 3 Flash Preview | 0.9125 (0.0225) | 0.6806 (0.0583) | -| 3.5 Flash | 0.9083 (0.0237) | 0.5602 (0.0444) | -| 3.6 Flash | 0.6042 (0.0195) | 0.6273 (0.0206) | -| 3.7 Flash | 0.5542 (0.0272) | 0.9000 (0.0340) | -| 3.8 Flash | 0.4875 (0.0209) | 0.9083 (0.0209) | +| release | x (SE) | y (SE) | status | +| --- | ---: | ---: | --- | +| 3 Flash Preview | 0.9125 (0.0225) | 0.6806 (0.0583) | preregistered panel | +| 3.5 Flash | 0.9083 (0.0237) | 0.5602 (0.0444) | preregistered panel | +| 3.6 Flash | 0.6042 (0.0195) | 0.6273 (0.0206) | preregistered panel | +| 3.7 Flash | 0.5542 (0.0272) | 0.9000 (0.0340) | smoke preregistered; panel preregistered | +| 3.8 Flash | 0.4875 (0.0209) | 0.9083 (0.0209) | exploratory, pre-prereg | -Release-date OLS: x slope -0.6062/year (SE 0.0496), y slope +0.2834/year (SE 0.0486), 2D residual RMSE 0.1541 (SE 0.0122). +All-items release OLS (exploratory, not a prediction test): x slope -0.6062/year (SE 0.0496), y slope +0.2834/year (SE 0.0486), 2D residual RMSE 0.1541 (SE 0.0122). + +## Fixed-item-set sensitivity (correction 2, the decisive comparison) + +The all-items comparison changed the estimand: original-choice 3.7/3.8 Y coordinates omit God and Abortion (zero coverage) while the dense comparators use the full item sets, and different releases omit different items. The corrected comparison fixes one item set per coverage rule, applied to ALL five original-choice releases simultaneously, and recomputes both dense `normal_minimum` and `normal_high` comparators on exactly the same items. Original-choice coverage thresholds are on the substantive-answer fraction; dense comparators have 6/6 valid samples on every item. + +| rule | X items | Y items | original RMSE | dense min RMSE | dense high RMSE | +| --- | ---: | ---: | ---: | ---: | ---: | +| common nonzero coverage | 5 | 5 | 0.1232 | 0.0627 | 0.0521 | +| >= 25% per-model coverage | 3 | 5 | 0.1691 | 0.1106 | 0.0937 | +| >= 50% per-model coverage | 1 | 5 | 0.2643 | 0.1259 | 0.1745 | +| >= 75% per-model coverage | 0 | 0 | undefined (no common items) | | | + +Item membership per rule (common nonzero): X = {Homosexuality, trust, petition, demonstrations, boycotts}; Y = {Religion, Obedience, Independence, Determination, Imagination}. cov25 drops Homosexuality and trust from X; cov50 keeps only Signing a petition in X. Full coordinates and slopes per rule are in `analysis.json` (`sensitivity_fixed_item_sets`). + +Under every rule where a comparison exists, the original-choice release-date 2D RMSE remains about 2.0-2.4x the dense minimum's and larger than the dense high's. The corrected estimand does not reverse the earlier direction; it weakens it (0.1541 all-items was inflated by the item-set mismatch) but the conclusion survives: the original single-choice format shows MORE release-date scatter, not less. The cov50 X axis is a single item, so its RMSE is that item's release variance and the comparison there is weakest. + +## Option-position and refusal contrast (correction 3, balance evidence) + +Rotation balance: every offered list is cyclically rotated by sample index, so each option occupies each rank exactly 24/k times by construction; bucket occupancy of the DK position across the 8 ordinary items per model is 110 first-third / 66 middle / 16 last-third (bucket edges, not the rotation, produce unequal bucket sizes). + +Refusal contrast (cannot_answer rate by the rank of the nearest non-substantive option, thirds): Preview 0.409/0.379/0.375, 3.5 0.073/0.106/0.063, 3.6 0.300/0.288/0.500, 3.7 0.591/0.621/0.750, 3.8 0.473/0.394/0.563 (first/middle/last). Selection share by position third is near-uniform for four of five models (0.30-0.36 per third); 3.7 mildly favors first-third (0.446). No contrast indicates the release-level patterns are a cyclic-order artifact: refusal rates differ strongly BETWEEN releases while staying nearly flat WITHIN release across DK positions. ## Comparison against the dense cells (descriptive only) -- Abstention: the original format does NOT reduce it. Overall `cannot_answer` rates (7.4%-54.6%, mean about 33%) overlap the dense flat-vector rates (35.3%-47.5% per cell, 40.7% overall), and the release heterogeneity is larger. Doubt migrates to the explicit non-substantive options rather than disappearing; P1 is supported. -- Scatter: the original-choice release-date 2D RMSE (0.1541, SE 0.0122) is about 2.4x-3.0x the dense cells' (normal minimum 0.0639, normal high 0.0518). P3 predicted no direction; observed direction is worse scatter. -- Coordinates differ from dense normal-minimum on most releases (P2 supported): for example Preview x 0.9125 vs 0.6170, and 3.7/3.8 y about 0.90 vs about 0.63. -- Slope directions also differ from every dense cell (dense x-slopes were between -0.237 and +0.135; here -0.606). +- Hidden versus explicit abstention: dense flat vectors are evidence of POSSIBLE hidden abstention, not observed refusals; only the original format's `cannot_answer` selections are observed abstentions. The two are related but not equivalent, and the dense flat rate (35.3-47.5% per cell, 40.7% overall) should not be read as a refusal rate. Under that caveat: explicit refusal in the original format (7.4-54.6%, mean about 33%) sits in the same range as the dense flat rates, so the original format does not clearly reduce total abstention; what it does is make part of it explicit and measurable (P1 supported). +- Scatter: on the fixed common item sets the original-choice release-date 2D RMSE is about 2.0-2.4x the dense cells' (see the sensitivity table). The corrected estimand confirms worse scatter for the original format. +- Coordinates differ from dense normal-minimum on most releases (P2 supported): for example Preview x 0.9125 vs 0.6170, and 3.7/3.8 y about 0.90 vs about 0.63 (all-items values; the fixed-set values are in `analysis.json`). +- Slope directions differ from the dense cells on the same fixed item sets (dense min x-slope -0.2369 on the common set vs original -0.6062). + +## Paid-call guard (correction 3) + +Root cause of the accidental run, precisely: `wvs_original_choice_pilot` imports `openrouter_request_with_metadata_once` via `from moralmaps.read_api import ...`, binding the function into the pilot module's namespace at import time. The pre-paid test monkeypatched `moralmaps.read_api.openrouter_request_with_metadata_once`, which rebinds only the read_api module attribute and never touches the pilot module's already-bound name, so the pilot kept calling the real API. Defenses added: (1) `budgeted_request` raises `PaidCallsNotAuthorized` unless env `WVS_PAID_CALLS_AUTHORIZED=1`, which only `--i-authorize-paid-calls` sets; (2) `--paid-smoke`/`--run` fail fast without that flag (the pueue wrapper passes it; offline tests do not); (3) CLI actions are mutually exclusive; (4) an offline regression (`--offline-regression`) proves no HTTP call executes without the opt-in even when the read_api level is patched, and that the CLI guard holds. ## Limits (unchanged from the dense pilot) diff --git a/slop/research/wvs/20260918_original_choice_pilot/analysis.json b/slop/research/wvs/20260918_original_choice_pilot/analysis.json index 9eed626..f2def90 100644 --- a/slop/research/wvs/20260918_original_choice_pilot/analysis.json +++ b/slop/research/wvs/20260918_original_choice_pilot/analysis.json @@ -120,7 +120,7 @@ "dealing with people?": 0.16666666666666666 } }, - "fits": { + "fits_all_items_exploratory": { "bootstrap": { "coord_xy_se": [ 0.022466332449912703, @@ -189,5 +189,507 @@ "slope_x_per_year": -0.6061984096087086, "slope_y_per_year": 0.2834030539477191 } + }, + "note": "the all-items five-release fit below is therefore NOT a preregistered prediction test; preregistered claims are per-release P1/P2 only", + "position_contrast": { + "google/gemini-3-flash-preview": { + "cannot_answer_rate_by_dk_position_third": { + "first_third": 0.4091, + "last_third": 0.375, + "middle": 0.3788 + }, + "n_by_dk_position_third": { + "first_third": 110, + "last_third": 16, + "middle": 66 + }, + "substantive_selection_share_by_position_third": { + "first_third": 0.3534, + "last_third": 0.3017, + "middle": 0.3448 + } + }, + "google/gemini-3.5-flash": { + "cannot_answer_rate_by_dk_position_third": { + "first_third": 0.0727, + "last_third": 0.0625, + "middle": 0.1061 + }, + "n_by_dk_position_third": { + "first_third": 110, + "last_third": 16, + "middle": 66 + }, + "substantive_selection_share_by_position_third": { + "first_third": 0.3352, + "last_third": 0.3068, + "middle": 0.358 + } + }, + "google/gemini-3.6-flash": { + "cannot_answer_rate_by_dk_position_third": { + "first_third": 0.3, + "last_third": 0.5, + "middle": 0.2879 + }, + "n_by_dk_position_third": { + "first_third": 110, + "last_third": 16, + "middle": 66 + }, + "substantive_selection_share_by_position_third": { + "first_third": 0.3409, + "last_third": 0.3182, + "middle": 0.3409 + } + }, + "google/gemini-3.7-flash": { + "cannot_answer_rate_by_dk_position_third": { + "first_third": 0.5909, + "last_third": 0.75, + "middle": 0.6212 + }, + "n_by_dk_position_third": { + "first_third": 110, + "last_third": 16, + "middle": 66 + }, + "substantive_selection_share_by_position_third": { + "first_third": 0.4459, + "last_third": 0.2432, + "middle": 0.3108 + } + }, + "google/gemini-3.8-flash": { + "cannot_answer_rate_by_dk_position_third": { + "first_third": 0.4727, + "last_third": 0.5625, + "middle": 0.3939 + }, + "n_by_dk_position_third": { + "first_third": 110, + "last_third": 16, + "middle": 66 + }, + "substantive_selection_share_by_position_third": { + "first_third": 0.3238, + "last_third": 0.3143, + "middle": 0.3619 + } + } + }, + "schema": 2, + "sensitivity_fixed_item_sets": { + "cov25": { + "dense_normal_high": { + "coords_xy_per_model": { + "google/gemini-3-flash-preview": [ + 0.40776014109347436, + 0.7365039281705948 + ], + "google/gemini-3.5-flash": [ + 0.5841049382716049, + 0.694352252685586 + ], + "google/gemini-3.6-flash": [ + 0.5435185185185184, + 0.6945206028539362 + ], + "google/gemini-3.7-flash": [ + 0.461089065255732, + 0.6669552669552671 + ], + "google/gemini-3.8-flash": [ + 0.3234126984126985, + 0.661552028218695 + ] + }, + "rmse_2d": 0.09374605926987896, + "slope_x_per_year": -0.012891078161264849, + "slope_y_per_year": -0.0989300777864456 + }, + "dense_normal_minimum": { + "coords_xy_per_model": { + "google/gemini-3-flash-preview": [ + 0.6434413580246913, + 0.6795634920634921 + ], + "google/gemini-3.5-flash": [ + 0.6374559082892416, + 0.624968734968735 + ], + "google/gemini-3.6-flash": [ + 0.6287037037037037, + 0.6437301587301588 + ], + "google/gemini-3.7-flash": [ + 0.38795194003527333, + 0.6824835658168993 + ], + "google/gemini-3.8-flash": [ + 0.2944003527336861, + 0.6769761103094437 + ] + }, + "rmse_2d": 0.11062522669412543, + "slope_x_per_year": -0.3918383833725489, + "slope_y_per_year": -0.0036253479507921704 + }, + "items": { + "Survival <-> Self-expression": [ + "Attending peaceful demonstrations", + "Joining in boycotts", + "Signing a petition" + ], + "Traditional <-> Secular-Rational": [ + "Determination, perseverance", + "Imagination", + "Independence", + "Obedience", + "Religion" + ] + }, + "min_per_model_coverage": 0.25, + "n_items": { + "Survival <-> Self-expression": 3, + "Traditional <-> Secular-Rational": 5 + }, + "original_choice": { + "coords_xy_per_model": { + "google/gemini-3-flash-preview": [ + 0.9375, + 0.8 + ], + "google/gemini-3.5-flash": [ + 0.9583333333333334, + 0.6805555555555556 + ], + "google/gemini-3.6-flash": [ + 0.7430555555555557, + 0.7444444444444445 + ], + "google/gemini-3.7-flash": [ + 0.5069444444444445, + 0.9 + ], + "google/gemini-3.8-flash": [ + 0.3541666666666667, + 0.9083333333333334 + ] + }, + "rmse_2d": 0.16910500932360994, + "slope_x_per_year": -0.7221633278414168, + "slope_y_per_year": 0.13230329671184193 + } + }, + "cov50": { + "dense_normal_high": { + "coords_xy_per_model": { + "google/gemini-3-flash-preview": [ + 0.5380952380952381, + 0.7365039281705948 + ], + "google/gemini-3.5-flash": [ + 0.7374338624338626, + 0.694352252685586 + ], + "google/gemini-3.6-flash": [ + 0.7, + 0.6945206028539362 + ], + "google/gemini-3.7-flash": [ + 0.5082671957671958, + 0.6669552669552671 + ], + "google/gemini-3.8-flash": [ + 0.22023809523809534, + 0.661552028218695 + ] + }, + "rmse_2d": 0.1745457540939421, + "slope_x_per_year": -0.2189997556490904, + "slope_y_per_year": -0.0989300777864456 + }, + "dense_normal_minimum": { + "coords_xy_per_model": { + "google/gemini-3-flash-preview": [ + 0.711111111111111, + 0.6795634920634921 + ], + "google/gemini-3.5-flash": [ + 0.7164351851851852, + 0.624968734968735 + ], + "google/gemini-3.6-flash": [ + 0.7185185185185186, + 0.6437301587301588 + ], + "google/gemini-3.7-flash": [ + 0.4495701058201058, + 0.6824835658168993 + ], + "google/gemini-3.8-flash": [ + 0.33558201058201065, + 0.6769761103094437 + ] + }, + "rmse_2d": 0.12592759141229765, + "slope_x_per_year": -0.40806802305431017, + "slope_y_per_year": -0.0036253479507921704 + }, + "items": { + "Survival <-> Self-expression": [ + "Signing a petition" + ], + "Traditional <-> Secular-Rational": [ + "Determination, perseverance", + "Imagination", + "Independence", + "Obedience", + "Religion" + ] + }, + "min_per_model_coverage": 0.5, + "n_items": { + "Survival <-> Self-expression": 1, + "Traditional <-> Secular-Rational": 5 + }, + "original_choice": { + "coords_xy_per_model": { + "google/gemini-3-flash-preview": [ + 0.9791666666666666, + 0.8 + ], + "google/gemini-3.5-flash": [ + 0.9791666666666666, + 0.6805555555555556 + ], + "google/gemini-3.6-flash": [ + 1.0, + 0.7444444444444445 + ], + "google/gemini-3.7-flash": [ + 0.33333333333333337, + 0.9 + ], + "google/gemini-3.8-flash": [ + 0.29166666666666663, + 0.9083333333333334 + ] + }, + "rmse_2d": 0.2643484950720592, + "slope_x_per_year": -0.8339691792184623, + "slope_y_per_year": 0.13230329671184193 + } + }, + "cov75": { + "dense_normal_high": { + "coords_xy_per_model": { + "google/gemini-3-flash-preview": [ + NaN, + 0.7365039281705948 + ], + "google/gemini-3.5-flash": [ + NaN, + 0.694352252685586 + ], + "google/gemini-3.6-flash": [ + NaN, + 0.6945206028539362 + ], + "google/gemini-3.7-flash": [ + NaN, + 0.6669552669552671 + ], + "google/gemini-3.8-flash": [ + NaN, + 0.661552028218695 + ] + }, + "rmse_2d": NaN, + "slope_x_per_year": NaN, + "slope_y_per_year": -0.0989300777864456 + }, + "dense_normal_minimum": { + "coords_xy_per_model": { + "google/gemini-3-flash-preview": [ + NaN, + 0.6795634920634921 + ], + "google/gemini-3.5-flash": [ + NaN, + 0.624968734968735 + ], + "google/gemini-3.6-flash": [ + NaN, + 0.6437301587301588 + ], + "google/gemini-3.7-flash": [ + NaN, + 0.6824835658168993 + ], + "google/gemini-3.8-flash": [ + NaN, + 0.6769761103094437 + ] + }, + "rmse_2d": NaN, + "slope_x_per_year": NaN, + "slope_y_per_year": -0.0036253479507921704 + }, + "items": { + "Survival <-> Self-expression": [], + "Traditional <-> Secular-Rational": [ + "Determination, perseverance", + "Imagination", + "Independence", + "Obedience", + "Religion" + ] + }, + "min_per_model_coverage": 0.75, + "n_items": { + "Survival <-> Self-expression": 0, + "Traditional <-> Secular-Rational": 5 + }, + "original_choice": { + "coords_xy_per_model": { + "google/gemini-3-flash-preview": [ + NaN, + 0.8 + ], + "google/gemini-3.5-flash": [ + NaN, + 0.6805555555555556 + ], + "google/gemini-3.6-flash": [ + NaN, + 0.7444444444444445 + ], + "google/gemini-3.7-flash": [ + NaN, + 0.9 + ], + "google/gemini-3.8-flash": [ + NaN, + 0.9083333333333334 + ] + }, + "rmse_2d": NaN, + "slope_x_per_year": NaN, + "slope_y_per_year": 0.13230329671184193 + } + }, + "nonzero": { + "dense_normal_high": { + "coords_xy_per_model": { + "google/gemini-3-flash-preview": [ + 0.4495943562610229, + 0.7365039281705948 + ], + "google/gemini-3.5-flash": [ + 0.5504629629629629, + 0.694352252685586 + ], + "google/gemini-3.6-flash": [ + 0.5522269705603039, + 0.6945206028539362 + ], + "google/gemini-3.7-flash": [ + 0.5077370426508357, + 0.6669552669552671 + ], + "google/gemini-3.8-flash": [ + 0.42397571564238234, + 0.661552028218695 + ] + }, + "rmse_2d": 0.05208403121789642, + "slope_x_per_year": 0.03264101067017449, + "slope_y_per_year": -0.0989300777864456 + }, + "dense_normal_minimum": { + "coords_xy_per_model": { + "google/gemini-3-flash-preview": [ + 0.6170217569786536, + 0.6795634920634921 + ], + "google/gemini-3.5-flash": [ + 0.6138227513227512, + 0.624968734968735 + ], + "google/gemini-3.6-flash": [ + 0.589126984126984, + 0.6437301587301588 + ], + "google/gemini-3.7-flash": [ + 0.4679434257738855, + 0.6824835658168993 + ], + "google/gemini-3.8-flash": [ + 0.41143769385148693, + 0.6769761103094437 + ] + }, + "rmse_2d": 0.06271183170463177, + "slope_x_per_year": -0.2368815371407787, + "slope_y_per_year": -0.0036253479507921704 + }, + "items": { + "Survival <-> Self-expression": [ + "Attending peaceful demonstrations", + "Homosexuality", + "Joining in boycotts", + "Signing a petition", + "dealing with people?" + ], + "Traditional <-> Secular-Rational": [ + "Determination, perseverance", + "Imagination", + "Independence", + "Obedience", + "Religion" + ] + }, + "min_per_model_coverage": 0.0, + "n_items": { + "Survival <-> Self-expression": 5, + "Traditional <-> Secular-Rational": 5 + }, + "original_choice": { + "coords_xy_per_model": { + "google/gemini-3-flash-preview": [ + 0.9125, + 0.8 + ], + "google/gemini-3.5-flash": [ + 0.9083333333333334, + 0.6805555555555556 + ], + "google/gemini-3.6-flash": [ + 0.6041666666666666, + 0.7444444444444445 + ], + "google/gemini-3.7-flash": [ + 0.5541666666666667, + 0.9 + ], + "google/gemini-3.8-flash": [ + 0.4875, + 0.9083333333333334 + ] + }, + "rmse_2d": 0.12317306101251788, + "slope_x_per_year": -0.6061984096087086, + "slope_y_per_year": 0.13230329671184193 + } + } + }, + "status_labels": { + "gemini-3-flash-preview": "preregistered", + "google/gemini-3.5-flash": "preregistered", + "google/gemini-3.6-flash": "preregistered", + "google/gemini-3.7-flash": "smoke-then-preregistered-panel", + "google/gemini-3.8-flash": "EXPLORATORY, ran before the preregistration commit 5ad68ad" } }