mirror of
https://github.com/wassname/moral-maps.git
synced 2026-10-06 13:10:36 +08:00
Open the think block for non-thinking-default templates, probe answer-slot readability
Qwen3.5 templates close an empty <think></think> by default, so the reader appended a second <think> and the model saw </think>...<think>. enable_thinking=True fixes the sequence; Qwen3 is unchanged (pmass 1.000 before and after). The probe exists because Qwen3.5-0.8B reads the WVS battery at pmass 0.61-0.84, not 0.999, and a mushy readout would waste the big run. Co-Authored-By: Claude <288921227+claudypoo@users.noreply.github.com>
This commit is contained in:
1 parent
3cd62abf2a
commit
9109dbfe19
6 files changed
+125
-9
No files matched your search
@@ -28,3 +28,7 @@ wvs-steer-modal model="Qwen/Qwen3.5-27B" gpu="H200":
|
||||
|
||||
wvs-steer-pull:
|
||||
uv run --group dev modal volume get --force moralmaps-wvs-steer outputs .
|
||||
|
||||
# which candidate large model reads the answer slot cleanly enough to be worth a sweep
|
||||
wvs-steer-readable models="Qwen/Qwen3.5-27B,Qwen/Qwen3-32B":
|
||||
uv run --extra steer --group dev modal run scripts/run_modal_wvs.py::readable --models {{models}}
|
||||
@@ -0,0 +1,70 @@
|
||||
"""Does the WVS answer-slot readout hold on this model family, and at what think budget?
|
||||
|
||||
Qwen3-0.6B answers the IW battery with pmass 0.999. Qwen3.5-0.8B read 0.783 at think=1 in the
|
||||
steering smoke, which would make every steered coordinate mushy. Before renting a big GPU, find out
|
||||
whether that is the think budget (the model is mid-thought when we force the answer slot) or the
|
||||
chat template (the prefill does not land where we think it does).
|
||||
|
||||
uv run --extra steer python scripts/probe_wvs_think_budget.py --model Qwen/Qwen3.5-0.8B
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
from loguru import logger
|
||||
from tabulate import tabulate
|
||||
from transformers import AutoModelForCausalLM, AutoTokenizer
|
||||
|
||||
from moralmaps.iw_axes import resolve_items
|
||||
from moralmaps.read import read_items, resolve_answer_ids
|
||||
from moralmaps.wvs import build_instruments, load_wvs_all, model_axis_scores, read_model
|
||||
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--model", default="Qwen/Qwen3.5-0.8B")
|
||||
ap.add_argument("--think-budgets", default="1,16,64,256")
|
||||
ap.add_argument("--batch-size", type=int, default=12)
|
||||
ap.add_argument("--device", default="cuda")
|
||||
ap.add_argument("--device-map", default=None, help="'auto' shards a large model over the GPUs")
|
||||
args = ap.parse_args()
|
||||
|
||||
tok = AutoTokenizer.from_pretrained(args.model)
|
||||
if tok.pad_token is None:
|
||||
tok.pad_token = tok.eos_token
|
||||
tok.padding_side = "left"
|
||||
if args.device_map:
|
||||
model = AutoModelForCausalLM.from_pretrained(args.model, dtype=torch.bfloat16,
|
||||
device_map=args.device_map).eval()
|
||||
else:
|
||||
model = AutoModelForCausalLM.from_pretrained(args.model, dtype=torch.bfloat16).to(args.device).eval()
|
||||
|
||||
resolved = resolve_items(load_wvs_all())
|
||||
instrs, meta = build_instruments(resolved)
|
||||
|
||||
rows = []
|
||||
for think in [int(t) for t in args.think_budgets.split(",")]:
|
||||
read = []
|
||||
for k, instr in enumerate(instrs):
|
||||
read += read_items(model, tok, instr, instr.items,
|
||||
resolve_answer_ids(tok, instr.answer_space),
|
||||
max_think_tokens=think, batch_size=args.batch_size,
|
||||
verbose_first=(k == 0 and think == 1))
|
||||
pm = [r["pmass_allowed"] for r in read]
|
||||
x, y = model_axis_scores(read_model(read, meta), meta, resolved)
|
||||
closed = sum(r["emitted_close"] for r in read)
|
||||
rows.append([think, f"{np.mean(pm):.3f}", f"{np.min(pm):.3f}", f"{closed}/{len(read)}",
|
||||
f"{x:.4f}", f"{y:.4f}"])
|
||||
logger.info(f"think={think}: mean pmass {np.mean(pm):.3f}")
|
||||
|
||||
print(tabulate(rows, tablefmt="pipe", headers=[
|
||||
"think tokens", "mean pmass", "min pmass", "closed think", "x", "y"]))
|
||||
print("\nSHOULD: pmass climbs toward ~1.0 as the think budget grows, and the coordinate settles.\n"
|
||||
"ELSE, if pmass stays low at every budget, the prefill or chat template is wrong for this\n"
|
||||
"family and no steered coordinate from it is comparable to the published map.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -83,6 +83,30 @@ def main(model: str = MODEL, methods: str = ",".join(METHODS), seeds: str = ",".
|
||||
print(f"{method}\ts{seed}\tFAILED\t{error}")
|
||||
|
||||
|
||||
@app.function(gpu=os.environ.get("WVS_GPU", "H200"), volumes={"/cache": cache}, timeout=60 * 60)
|
||||
def probe(model: str, device_map: str) -> str:
|
||||
"""Read the battery unsteered at several think budgets: is this model's answer slot readable?"""
|
||||
from huggingface_hub import snapshot_download
|
||||
|
||||
snapshot_download(model)
|
||||
argv = ["--model", model] + (["--device-map", device_map] if device_map else [])
|
||||
out = subprocess.run([sys.executable, "scripts/probe_wvs_think_budget.py", *argv],
|
||||
cwd="/repo", check=True, capture_output=True, text=True)
|
||||
return out.stdout
|
||||
|
||||
|
||||
@app.local_entrypoint()
|
||||
def readable(models: str = "Qwen/Qwen3.5-27B,Qwen/Qwen3-32B", device_map: str = ""):
|
||||
"""Which candidate large model reads cleanly enough to be worth a sweep? ~10 min per model."""
|
||||
handles = {m: probe.spawn(m, device_map) for m in models.split(",")}
|
||||
for m, handle in handles.items():
|
||||
print(f"\n===== {m} =====")
|
||||
try:
|
||||
print(handle.get())
|
||||
except Exception as error:
|
||||
print(f"FAILED\t{error}")
|
||||
|
||||
|
||||
@app.local_entrypoint()
|
||||
def smoke():
|
||||
"""Same image, mounts and Volume as the real fan-out, on the tiny random model."""
|
||||
|
||||
@@ -141,7 +141,9 @@ def main() -> None:
|
||||
else:
|
||||
model = AutoModelForCausalLM.from_pretrained(args.model, dtype=dtype).to(args.device).eval()
|
||||
|
||||
n_blocks = model.config.num_hidden_layers
|
||||
# Qwen3.5 ships as a VL wrapper (Qwen3_5ForConditionalGeneration), so the block count lives in
|
||||
# config.text_config, not at the top level.
|
||||
n_blocks = model.config.get_text_config().num_hidden_layers
|
||||
if args.layers == "mid":
|
||||
layers = tuple(range(max(2, int(n_blocks * 0.2)), min(n_blocks - 2, int(n_blocks * 0.8))))
|
||||
else:
|
||||
|
||||
@@ -7,6 +7,15 @@ written by Claude (claude-opus-4.8 in pi), 2026-09-18. NOT yet approved by wassn
|
||||
> moral map how the cultural preference change when steered for honesty + credulity. I want to
|
||||
> try a few steering methods." -- wassname
|
||||
|
||||
Decisions after the first pass (wassname, 2026-09-18):
|
||||
- honesty only, one axis, credulity dropped
|
||||
- WVS only, no MFQ-2. "that's the plot"
|
||||
- "$100 is fine if it works"
|
||||
|
||||
Branch note: this worktree branched from 789829e, which the Gemini-pilot session later amended to
|
||||
4955468. Rebase or cherry-pick the steering commits onto 4955468 before any merge, do not carry the
|
||||
stale preregistration commit. (told to me over intercom by moralmap_add_astra)
|
||||
|
||||
## What exists already (observation, read from the code today)
|
||||
|
||||
- `scripts/wvs_map.py --local-model X` already reads the 12-item Inglehart-Welzel battery with the
|
||||
@@ -51,13 +60,15 @@ Why not the biggest: Kimi K3 does not fit one Modal container at any precision,
|
||||
fits as fp8 where the `vjp_delta` backward is a research project of its own. Qwen3.5-122B-A10B also
|
||||
gives a clean scaling story next to the published Qwen3-4B showcase, same family.
|
||||
|
||||
## Preregistered confound (raise before running)
|
||||
## Preregistered confound
|
||||
|
||||
X-axis item 2 is interpersonal trust, `"Most people can be trusted"`. A **credulity** steer is
|
||||
close to a paraphrase of that item, so an X shift may be lexical leakage, not a cultural move.
|
||||
Discriminator: recompute X with that item held out. If the credulity effect on X survives the
|
||||
holdout, it is a cultural move; if X collapses to base, it is item leakage and must be reported
|
||||
as such. Honesty has no equivalent overlap in the battery.
|
||||
The battery is 12 items, so one item can carry a whole apparent move. `scripts/wvs_steer_sweep.py`
|
||||
saves each item's position at every dose, and `scripts/plot_wvs_steer.py` reports the move length
|
||||
again with the single most influential item removed. A move that mostly disappears under that
|
||||
leave-one-out is one item reacting to the persona wording, not a cultural shift.
|
||||
|
||||
This mattered most for credulity, which nearly paraphrases the X-axis trust item
|
||||
(`"Most people can be trusted"`). Credulity is now out of scope, but the check is cheap and stays.
|
||||
|
||||
## Goals
|
||||
|
||||
|
||||
@@ -27,9 +27,14 @@ _ASSISTANT_SENTINEL: str = "ZZUNIQ_ASSISTANT_SENTINEL_ZZ"
|
||||
|
||||
|
||||
def _generation_prompt_with_open_think(tok, messages: list[dict[str, str]]) -> str:
|
||||
"""Return exactly one open reasoning marker. (Claude, 2026-07-19)"""
|
||||
"""Return exactly one open reasoning marker. (Claude, 2026-07-19)
|
||||
|
||||
enable_thinking=True matters for templates that default to non-thinking: Qwen3.5 otherwise emits
|
||||
a CLOSED empty `<think></think>`, we then append our own `<think>`, and the model reads a
|
||||
`</think> ... <think>` sequence it never saw in training. Templates without the variable ignore
|
||||
it (Qwen3 is unchanged)."""
|
||||
prompt = tok.apply_chat_template(
|
||||
messages, tokenize=False, add_generation_prompt=True)
|
||||
messages, tokenize=False, add_generation_prompt=True, enable_thinking=True)
|
||||
if prompt.rstrip().endswith("<think>"):
|
||||
return prompt
|
||||
return prompt + "<think>\n"
|
||||
|
||||
Reference in new issue
Block a user