From 9109dbfe19ff7a576fc089ff2f665e359a0132d9 Mon Sep 17 00:00:00 2001
From: wassname <1103714+wassname@users.noreply.github.com>
Date: Fri, 18 Sep 2026 19:31:58 +0800
Subject: [PATCH] Open the think block for non-thinking-default templates,
probe answer-slot readability
Qwen3.5 templates close an empty by default, so the reader appended a
second and the model saw .... enable_thinking=True fixes the
sequence; Qwen3 is unchanged (pmass 1.000 before and after).
The probe exists because Qwen3.5-0.8B reads the WVS battery at pmass 0.61-0.84, not
0.999, and a mushy readout would waste the big run.
Co-Authored-By: Claude <288921227+claudypoo@users.noreply.github.com>
---
justfile | 4 ++
scripts/probe_wvs_think_budget.py | 70 ++++++++++++++++++++++++++++
scripts/run_modal_wvs.py | 24 ++++++++++
scripts/wvs_steer_sweep.py | 4 +-
slop/plans/20260918_wvs_steer_big.md | 23 ++++++---
src/moralmaps/guided.py | 9 +++-
6 files changed, 125 insertions(+), 9 deletions(-)
create mode 100644 scripts/probe_wvs_think_budget.py
diff --git a/justfile b/justfile
index 3f4818f..c6a6e0d 100644
--- a/justfile
+++ b/justfile
@@ -28,3 +28,7 @@ wvs-steer-modal model="Qwen/Qwen3.5-27B" gpu="H200":
wvs-steer-pull:
uv run --group dev modal volume get --force moralmaps-wvs-steer outputs .
+
+# which candidate large model reads the answer slot cleanly enough to be worth a sweep
+wvs-steer-readable models="Qwen/Qwen3.5-27B,Qwen/Qwen3-32B":
+ uv run --extra steer --group dev modal run scripts/run_modal_wvs.py::readable --models {{models}}
diff --git a/scripts/probe_wvs_think_budget.py b/scripts/probe_wvs_think_budget.py
new file mode 100644
index 0000000..28674b4
--- /dev/null
+++ b/scripts/probe_wvs_think_budget.py
@@ -0,0 +1,70 @@
+"""Does the WVS answer-slot readout hold on this model family, and at what think budget?
+
+Qwen3-0.6B answers the IW battery with pmass 0.999. Qwen3.5-0.8B read 0.783 at think=1 in the
+steering smoke, which would make every steered coordinate mushy. Before renting a big GPU, find out
+whether that is the think budget (the model is mid-thought when we force the answer slot) or the
+chat template (the prefill does not land where we think it does).
+
+ uv run --extra steer python scripts/probe_wvs_think_budget.py --model Qwen/Qwen3.5-0.8B
+"""
+from __future__ import annotations
+
+import argparse
+
+import numpy as np
+import torch
+from loguru import logger
+from tabulate import tabulate
+from transformers import AutoModelForCausalLM, AutoTokenizer
+
+from moralmaps.iw_axes import resolve_items
+from moralmaps.read import read_items, resolve_answer_ids
+from moralmaps.wvs import build_instruments, load_wvs_all, model_axis_scores, read_model
+
+
+def main() -> None:
+ ap = argparse.ArgumentParser()
+ ap.add_argument("--model", default="Qwen/Qwen3.5-0.8B")
+ ap.add_argument("--think-budgets", default="1,16,64,256")
+ ap.add_argument("--batch-size", type=int, default=12)
+ ap.add_argument("--device", default="cuda")
+ ap.add_argument("--device-map", default=None, help="'auto' shards a large model over the GPUs")
+ args = ap.parse_args()
+
+ tok = AutoTokenizer.from_pretrained(args.model)
+ if tok.pad_token is None:
+ tok.pad_token = tok.eos_token
+ tok.padding_side = "left"
+ if args.device_map:
+ model = AutoModelForCausalLM.from_pretrained(args.model, dtype=torch.bfloat16,
+ device_map=args.device_map).eval()
+ else:
+ model = AutoModelForCausalLM.from_pretrained(args.model, dtype=torch.bfloat16).to(args.device).eval()
+
+ resolved = resolve_items(load_wvs_all())
+ instrs, meta = build_instruments(resolved)
+
+ rows = []
+ for think in [int(t) for t in args.think_budgets.split(",")]:
+ read = []
+ for k, instr in enumerate(instrs):
+ read += read_items(model, tok, instr, instr.items,
+ resolve_answer_ids(tok, instr.answer_space),
+ max_think_tokens=think, batch_size=args.batch_size,
+ verbose_first=(k == 0 and think == 1))
+ pm = [r["pmass_allowed"] for r in read]
+ x, y = model_axis_scores(read_model(read, meta), meta, resolved)
+ closed = sum(r["emitted_close"] for r in read)
+ rows.append([think, f"{np.mean(pm):.3f}", f"{np.min(pm):.3f}", f"{closed}/{len(read)}",
+ f"{x:.4f}", f"{y:.4f}"])
+ logger.info(f"think={think}: mean pmass {np.mean(pm):.3f}")
+
+ print(tabulate(rows, tablefmt="pipe", headers=[
+ "think tokens", "mean pmass", "min pmass", "closed think", "x", "y"]))
+ print("\nSHOULD: pmass climbs toward ~1.0 as the think budget grows, and the coordinate settles.\n"
+ "ELSE, if pmass stays low at every budget, the prefill or chat template is wrong for this\n"
+ "family and no steered coordinate from it is comparable to the published map.")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/run_modal_wvs.py b/scripts/run_modal_wvs.py
index add7f26..46a72c4 100644
--- a/scripts/run_modal_wvs.py
+++ b/scripts/run_modal_wvs.py
@@ -83,6 +83,30 @@ def main(model: str = MODEL, methods: str = ",".join(METHODS), seeds: str = ",".
print(f"{method}\ts{seed}\tFAILED\t{error}")
+@app.function(gpu=os.environ.get("WVS_GPU", "H200"), volumes={"/cache": cache}, timeout=60 * 60)
+def probe(model: str, device_map: str) -> str:
+ """Read the battery unsteered at several think budgets: is this model's answer slot readable?"""
+ from huggingface_hub import snapshot_download
+
+ snapshot_download(model)
+ argv = ["--model", model] + (["--device-map", device_map] if device_map else [])
+ out = subprocess.run([sys.executable, "scripts/probe_wvs_think_budget.py", *argv],
+ cwd="/repo", check=True, capture_output=True, text=True)
+ return out.stdout
+
+
+@app.local_entrypoint()
+def readable(models: str = "Qwen/Qwen3.5-27B,Qwen/Qwen3-32B", device_map: str = ""):
+ """Which candidate large model reads cleanly enough to be worth a sweep? ~10 min per model."""
+ handles = {m: probe.spawn(m, device_map) for m in models.split(",")}
+ for m, handle in handles.items():
+ print(f"\n===== {m} =====")
+ try:
+ print(handle.get())
+ except Exception as error:
+ print(f"FAILED\t{error}")
+
+
@app.local_entrypoint()
def smoke():
"""Same image, mounts and Volume as the real fan-out, on the tiny random model."""
diff --git a/scripts/wvs_steer_sweep.py b/scripts/wvs_steer_sweep.py
index 1a272ab..dae1fd7 100644
--- a/scripts/wvs_steer_sweep.py
+++ b/scripts/wvs_steer_sweep.py
@@ -141,7 +141,9 @@ def main() -> None:
else:
model = AutoModelForCausalLM.from_pretrained(args.model, dtype=dtype).to(args.device).eval()
- n_blocks = model.config.num_hidden_layers
+ # Qwen3.5 ships as a VL wrapper (Qwen3_5ForConditionalGeneration), so the block count lives in
+ # config.text_config, not at the top level.
+ n_blocks = model.config.get_text_config().num_hidden_layers
if args.layers == "mid":
layers = tuple(range(max(2, int(n_blocks * 0.2)), min(n_blocks - 2, int(n_blocks * 0.8))))
else:
diff --git a/slop/plans/20260918_wvs_steer_big.md b/slop/plans/20260918_wvs_steer_big.md
index e78ae8c..c7da2c2 100644
--- a/slop/plans/20260918_wvs_steer_big.md
+++ b/slop/plans/20260918_wvs_steer_big.md
@@ -7,6 +7,15 @@ written by Claude (claude-opus-4.8 in pi), 2026-09-18. NOT yet approved by wassn
> moral map how the cultural preference change when steered for honesty + credulity. I want to
> try a few steering methods." -- wassname
+Decisions after the first pass (wassname, 2026-09-18):
+- honesty only, one axis, credulity dropped
+- WVS only, no MFQ-2. "that's the plot"
+- "$100 is fine if it works"
+
+Branch note: this worktree branched from 789829e, which the Gemini-pilot session later amended to
+4955468. Rebase or cherry-pick the steering commits onto 4955468 before any merge, do not carry the
+stale preregistration commit. (told to me over intercom by moralmap_add_astra)
+
## What exists already (observation, read from the code today)
- `scripts/wvs_map.py --local-model X` already reads the 12-item Inglehart-Welzel battery with the
@@ -51,13 +60,15 @@ Why not the biggest: Kimi K3 does not fit one Modal container at any precision,
fits as fp8 where the `vjp_delta` backward is a research project of its own. Qwen3.5-122B-A10B also
gives a clean scaling story next to the published Qwen3-4B showcase, same family.
-## Preregistered confound (raise before running)
+## Preregistered confound
-X-axis item 2 is interpersonal trust, `"Most people can be trusted"`. A **credulity** steer is
-close to a paraphrase of that item, so an X shift may be lexical leakage, not a cultural move.
-Discriminator: recompute X with that item held out. If the credulity effect on X survives the
-holdout, it is a cultural move; if X collapses to base, it is item leakage and must be reported
-as such. Honesty has no equivalent overlap in the battery.
+The battery is 12 items, so one item can carry a whole apparent move. `scripts/wvs_steer_sweep.py`
+saves each item's position at every dose, and `scripts/plot_wvs_steer.py` reports the move length
+again with the single most influential item removed. A move that mostly disappears under that
+leave-one-out is one item reacting to the persona wording, not a cultural shift.
+
+This mattered most for credulity, which nearly paraphrases the X-axis trust item
+(`"Most people can be trusted"`). Credulity is now out of scope, but the check is cheap and stays.
## Goals
diff --git a/src/moralmaps/guided.py b/src/moralmaps/guided.py
index bae0f52..a62dced 100644
--- a/src/moralmaps/guided.py
+++ b/src/moralmaps/guided.py
@@ -27,9 +27,14 @@ _ASSISTANT_SENTINEL: str = "ZZUNIQ_ASSISTANT_SENTINEL_ZZ"
def _generation_prompt_with_open_think(tok, messages: list[dict[str, str]]) -> str:
- """Return exactly one open reasoning marker. (Claude, 2026-07-19)"""
+ """Return exactly one open reasoning marker. (Claude, 2026-07-19)
+
+ enable_thinking=True matters for templates that default to non-thinking: Qwen3.5 otherwise emits
+ a CLOSED empty ``, we then append our own ``, and the model reads a
+ ` ... ` sequence it never saw in training. Templates without the variable ignore
+ it (Qwen3 is unchanged)."""
prompt = tok.apply_chat_template(
- messages, tokenize=False, add_generation_prompt=True)
+ messages, tokenize=False, add_generation_prompt=True, enable_thinking=True)
if prompt.rstrip().endswith(""):
return prompt
return prompt + "\n"