diff --git a/justfile b/justfile index 3f4818f..c6a6e0d 100644 --- a/justfile +++ b/justfile @@ -28,3 +28,7 @@ wvs-steer-modal model="Qwen/Qwen3.5-27B" gpu="H200": wvs-steer-pull: uv run --group dev modal volume get --force moralmaps-wvs-steer outputs . + +# which candidate large model reads the answer slot cleanly enough to be worth a sweep +wvs-steer-readable models="Qwen/Qwen3.5-27B,Qwen/Qwen3-32B": + uv run --extra steer --group dev modal run scripts/run_modal_wvs.py::readable --models {{models}} diff --git a/scripts/probe_wvs_think_budget.py b/scripts/probe_wvs_think_budget.py new file mode 100644 index 0000000..28674b4 --- /dev/null +++ b/scripts/probe_wvs_think_budget.py @@ -0,0 +1,70 @@ +"""Does the WVS answer-slot readout hold on this model family, and at what think budget? + +Qwen3-0.6B answers the IW battery with pmass 0.999. Qwen3.5-0.8B read 0.783 at think=1 in the +steering smoke, which would make every steered coordinate mushy. Before renting a big GPU, find out +whether that is the think budget (the model is mid-thought when we force the answer slot) or the +chat template (the prefill does not land where we think it does). + + uv run --extra steer python scripts/probe_wvs_think_budget.py --model Qwen/Qwen3.5-0.8B +""" +from __future__ import annotations + +import argparse + +import numpy as np +import torch +from loguru import logger +from tabulate import tabulate +from transformers import AutoModelForCausalLM, AutoTokenizer + +from moralmaps.iw_axes import resolve_items +from moralmaps.read import read_items, resolve_answer_ids +from moralmaps.wvs import build_instruments, load_wvs_all, model_axis_scores, read_model + + +def main() -> None: + ap = argparse.ArgumentParser() + ap.add_argument("--model", default="Qwen/Qwen3.5-0.8B") + ap.add_argument("--think-budgets", default="1,16,64,256") + ap.add_argument("--batch-size", type=int, default=12) + ap.add_argument("--device", default="cuda") + ap.add_argument("--device-map", default=None, help="'auto' shards a large model over the GPUs") + args = ap.parse_args() + + tok = AutoTokenizer.from_pretrained(args.model) + if tok.pad_token is None: + tok.pad_token = tok.eos_token + tok.padding_side = "left" + if args.device_map: + model = AutoModelForCausalLM.from_pretrained(args.model, dtype=torch.bfloat16, + device_map=args.device_map).eval() + else: + model = AutoModelForCausalLM.from_pretrained(args.model, dtype=torch.bfloat16).to(args.device).eval() + + resolved = resolve_items(load_wvs_all()) + instrs, meta = build_instruments(resolved) + + rows = [] + for think in [int(t) for t in args.think_budgets.split(",")]: + read = [] + for k, instr in enumerate(instrs): + read += read_items(model, tok, instr, instr.items, + resolve_answer_ids(tok, instr.answer_space), + max_think_tokens=think, batch_size=args.batch_size, + verbose_first=(k == 0 and think == 1)) + pm = [r["pmass_allowed"] for r in read] + x, y = model_axis_scores(read_model(read, meta), meta, resolved) + closed = sum(r["emitted_close"] for r in read) + rows.append([think, f"{np.mean(pm):.3f}", f"{np.min(pm):.3f}", f"{closed}/{len(read)}", + f"{x:.4f}", f"{y:.4f}"]) + logger.info(f"think={think}: mean pmass {np.mean(pm):.3f}") + + print(tabulate(rows, tablefmt="pipe", headers=[ + "think tokens", "mean pmass", "min pmass", "closed think", "x", "y"])) + print("\nSHOULD: pmass climbs toward ~1.0 as the think budget grows, and the coordinate settles.\n" + "ELSE, if pmass stays low at every budget, the prefill or chat template is wrong for this\n" + "family and no steered coordinate from it is comparable to the published map.") + + +if __name__ == "__main__": + main() diff --git a/scripts/run_modal_wvs.py b/scripts/run_modal_wvs.py index add7f26..46a72c4 100644 --- a/scripts/run_modal_wvs.py +++ b/scripts/run_modal_wvs.py @@ -83,6 +83,30 @@ def main(model: str = MODEL, methods: str = ",".join(METHODS), seeds: str = ",". print(f"{method}\ts{seed}\tFAILED\t{error}") +@app.function(gpu=os.environ.get("WVS_GPU", "H200"), volumes={"/cache": cache}, timeout=60 * 60) +def probe(model: str, device_map: str) -> str: + """Read the battery unsteered at several think budgets: is this model's answer slot readable?""" + from huggingface_hub import snapshot_download + + snapshot_download(model) + argv = ["--model", model] + (["--device-map", device_map] if device_map else []) + out = subprocess.run([sys.executable, "scripts/probe_wvs_think_budget.py", *argv], + cwd="/repo", check=True, capture_output=True, text=True) + return out.stdout + + +@app.local_entrypoint() +def readable(models: str = "Qwen/Qwen3.5-27B,Qwen/Qwen3-32B", device_map: str = ""): + """Which candidate large model reads cleanly enough to be worth a sweep? ~10 min per model.""" + handles = {m: probe.spawn(m, device_map) for m in models.split(",")} + for m, handle in handles.items(): + print(f"\n===== {m} =====") + try: + print(handle.get()) + except Exception as error: + print(f"FAILED\t{error}") + + @app.local_entrypoint() def smoke(): """Same image, mounts and Volume as the real fan-out, on the tiny random model.""" diff --git a/scripts/wvs_steer_sweep.py b/scripts/wvs_steer_sweep.py index 1a272ab..dae1fd7 100644 --- a/scripts/wvs_steer_sweep.py +++ b/scripts/wvs_steer_sweep.py @@ -141,7 +141,9 @@ def main() -> None: else: model = AutoModelForCausalLM.from_pretrained(args.model, dtype=dtype).to(args.device).eval() - n_blocks = model.config.num_hidden_layers + # Qwen3.5 ships as a VL wrapper (Qwen3_5ForConditionalGeneration), so the block count lives in + # config.text_config, not at the top level. + n_blocks = model.config.get_text_config().num_hidden_layers if args.layers == "mid": layers = tuple(range(max(2, int(n_blocks * 0.2)), min(n_blocks - 2, int(n_blocks * 0.8)))) else: diff --git a/slop/plans/20260918_wvs_steer_big.md b/slop/plans/20260918_wvs_steer_big.md index e78ae8c..c7da2c2 100644 --- a/slop/plans/20260918_wvs_steer_big.md +++ b/slop/plans/20260918_wvs_steer_big.md @@ -7,6 +7,15 @@ written by Claude (claude-opus-4.8 in pi), 2026-09-18. NOT yet approved by wassn > moral map how the cultural preference change when steered for honesty + credulity. I want to > try a few steering methods." -- wassname +Decisions after the first pass (wassname, 2026-09-18): +- honesty only, one axis, credulity dropped +- WVS only, no MFQ-2. "that's the plot" +- "$100 is fine if it works" + +Branch note: this worktree branched from 789829e, which the Gemini-pilot session later amended to +4955468. Rebase or cherry-pick the steering commits onto 4955468 before any merge, do not carry the +stale preregistration commit. (told to me over intercom by moralmap_add_astra) + ## What exists already (observation, read from the code today) - `scripts/wvs_map.py --local-model X` already reads the 12-item Inglehart-Welzel battery with the @@ -51,13 +60,15 @@ Why not the biggest: Kimi K3 does not fit one Modal container at any precision, fits as fp8 where the `vjp_delta` backward is a research project of its own. Qwen3.5-122B-A10B also gives a clean scaling story next to the published Qwen3-4B showcase, same family. -## Preregistered confound (raise before running) +## Preregistered confound -X-axis item 2 is interpersonal trust, `"Most people can be trusted"`. A **credulity** steer is -close to a paraphrase of that item, so an X shift may be lexical leakage, not a cultural move. -Discriminator: recompute X with that item held out. If the credulity effect on X survives the -holdout, it is a cultural move; if X collapses to base, it is item leakage and must be reported -as such. Honesty has no equivalent overlap in the battery. +The battery is 12 items, so one item can carry a whole apparent move. `scripts/wvs_steer_sweep.py` +saves each item's position at every dose, and `scripts/plot_wvs_steer.py` reports the move length +again with the single most influential item removed. A move that mostly disappears under that +leave-one-out is one item reacting to the persona wording, not a cultural shift. + +This mattered most for credulity, which nearly paraphrases the X-axis trust item +(`"Most people can be trusted"`). Credulity is now out of scope, but the check is cheap and stays. ## Goals diff --git a/src/moralmaps/guided.py b/src/moralmaps/guided.py index bae0f52..a62dced 100644 --- a/src/moralmaps/guided.py +++ b/src/moralmaps/guided.py @@ -27,9 +27,14 @@ _ASSISTANT_SENTINEL: str = "ZZUNIQ_ASSISTANT_SENTINEL_ZZ" def _generation_prompt_with_open_think(tok, messages: list[dict[str, str]]) -> str: - """Return exactly one open reasoning marker. (Claude, 2026-07-19)""" + """Return exactly one open reasoning marker. (Claude, 2026-07-19) + + enable_thinking=True matters for templates that default to non-thinking: Qwen3.5 otherwise emits + a CLOSED empty ``, we then append our own ``, and the model reads a + ` ... ` sequence it never saw in training. Templates without the variable ignore + it (Qwen3 is unchanged).""" prompt = tok.apply_chat_template( - messages, tokenize=False, add_generation_prompt=True) + messages, tokenize=False, add_generation_prompt=True, enable_thinking=True) if prompt.rstrip().endswith(""): return prompt return prompt + "\n"