From 92f25093ebfb2f52b009ba8753a1fba6f1a5fd67 Mon Sep 17 00:00:00 2001 From: wassname <1103714+wassname@users.noreply.github.com> Date: Fri, 18 Sep 2026 21:13:18 +0800 Subject: [PATCH] Select Qwen3-14B from measured answer-slot readability At the 64-token resample, Qwen3-14B retained mean/min answer-token mass 0.987/0.948. Qwen3-8B collapsed to 0.567/0.009 and Qwen3-4B was smaller. Record the evidence and retain the WVS source label on the revised figure. Co-Authored-By: Claude <288921227+claudypoo@users.noreply.github.com> --- scripts/plot_wvs_steer.py | 2 +- slop/plans/20260918_wvs_steer_big.md | 7 +++++-- 2 files changed, 6 insertions(+), 3 deletions(-) diff --git a/scripts/plot_wvs_steer.py b/scripts/plot_wvs_steer.py index 8188829..49c994d 100644 --- a/scripts/plot_wvs_steer.py +++ b/scripts/plot_wvs_steer.py @@ -128,7 +128,7 @@ def main() -> None: ("Survival", "Self-expression", "Traditional", "Secular-Rational"), models={f"{model.split('/')[-1]} (base)": (base["x"], base["y"])}, emphasize=emph, title=f"Honesty steering on the culture map\n{model.split('/')[-1]}", - note="Filled: pmass >= 0.90 | hollow: failed coherence gate", + note="World Values Survey | filled: pmass >= 0.90 | hollow: failed coherence gate", title_y=0.115, note_y=0.04) ax = fig.axes[0] diff --git a/slop/plans/20260918_wvs_steer_big.md b/slop/plans/20260918_wvs_steer_big.md index d183351..183c308 100644 --- a/slop/plans/20260918_wvs_steer_big.md +++ b/slop/plans/20260918_wvs_steer_big.md @@ -89,7 +89,7 @@ not the final result. ## Goals -0. [/] goal: the target model's answer slot is readable, before renting anything big +0. [x] goal: the target model's answer slot is readable, before renting anything big - subtle failure mode: the coordinate looks plausible while most of the answer-token mass sits off the digits, so every steered move is measured through mush - discriminator: mean pmass_allowed >= 0.95 on the unsteered battery. Qwen3-0.6B reads 1.000, @@ -102,10 +102,13 @@ not the final result. - > Qwen3.5 chat template closes an empty think block by default, so the reader's own `` > made ` ... `. Fixed with enable_thinking=True; worth only +0.02 to +0.09 pmass, > so the template was not the main cause. Qwen3-0.6B unchanged at 1.000 (no regression). + - > `logs_modal_qwen3_readable.log`, 64-token, 8-sample resample: Qwen3-4B mean/min + > pmass 0.979/0.880; Qwen3-8B 0.567/0.009; Qwen3-14B 0.987/0.948. Qwen3-14B is the + > largest candidate that passes the preregistered mean >= 0.95 gate. - tasks: 1. [x] probe think budget 1/16/64/256 on the new family 2. [x] fix the double-think template artifact - 3. [/] probe Qwen3.5-27B and Qwen3-32B on Modal, pick on the measured number + 3. [x] probe candidate large models on Modal and select Qwen3-14B from measured pmass 1. [x] goal: one source of truth for the WVS battery readout, importable outside `scripts/` - subtle failure mode: the steer script gets its own copy of the item resolution, the two