Refresh canonical WVS map integration

Co-Authored-By: PI[k3] <288921227+claudypoo@users.noreply.github.com>
This commit is contained in:
wassname
2026-09-18 01:52:07 +08:00
co-authored by PI[k3]
parent 39c9a5ce98
commit 3aa2a09e49
15 changed files with 6058 additions and 4799 deletions
File diff suppressed because one or more lines are too long
Binary file not shown.

Before

Width:  |  Height:  |  Size: 404 KiB

After

Width:  |  Height:  |  Size: 444 KiB

+4980 -4445
View File
File diff suppressed because it is too large Load Diff

Before

Width:  |  Height:  |  Size: 493 KiB

After

Width:  |  Height:  |  Size: 515 KiB

+110 -66
View File
@@ -1,66 +1,110 @@
| model | x self-expr | y secular | x 95%CI | y 95%CI |
|:-------------------------------------|--------------:|------------:|----------:|----------:|
| qwen3.6-27b (rated) | +0.46 | +0.68 | +0.18 | +0.14 |
| qwen3.5-35b-a3b (rated) | +0.54 | +0.55 | +0.14 | +0.18 |
| qwen3.6-max-preview (rated) | +0.48 | +0.69 | +0.16 | +0.14 |
| muse-spark-1.3 (rated) | +0.43 | +0.73 | +0.18 | +0.12 |
| qwen3.8-27b (rated) | +0.43 | +0.69 | +0.16 | +0.14 |
| qwen3.8-flash (rated) | +0.42 | +0.63 | +0.14 | +0.15 |
| gemma-4-31b-it (rated) | +0.36 | +0.66 | +0.16 | +0.13 |
| qwen3.5-122b-a10b (rated) | +0.48 | +0.62 | +0.15 | +0.14 |
| grok-4.20 (rated) | +0.59 | +0.62 | +0.11 | +0.17 |
| qwen3.5-plus-20260420 (rated) | +0.59 | +0.65 | +0.12 | +0.16 |
| gemini-2.5-pro (rated) | +0.47 | +0.70 | +0.13 | +0.14 |
| grok-4.3 (rated) | +0.44 | +0.73 | +0.14 | +0.13 |
| deepseek-v4-flash (rated) | +0.57 | +0.64 | +0.11 | +0.16 |
| gpt-5.4 (rated) | +0.45 | +0.68 | +0.13 | +0.14 |
| deepseek-v4.1-flash (rated) | +0.54 | +0.60 | +0.10 | +0.16 |
| qwen3-coder-30b-a3b-instruct (rated) | +0.50 | +0.58 | +0.09 | +0.17 |
| qwen3-30b-a3b (rated) | +0.67 | +0.61 | +0.07 | +0.19 |
| gpt-6-astra (rated) | +0.46 | +0.68 | +0.15 | +0.11 |
| mistral-large-2512 (rated) | +0.65 | +0.64 | +0.08 | +0.18 |
| gemma-3-27b-it (rated) | +0.60 | +0.60 | +0.09 | +0.17 |
| llama-4-maverick (rated) | +0.64 | +0.64 | +0.08 | +0.18 |
| qwen3-coder (rated) | +0.60 | +0.62 | +0.08 | +0.18 |
| qwen3.7-flash (rated) | +0.65 | +0.59 | +0.10 | +0.16 |
| qwen3-vl-235b-a22b-instruct (rated) | +0.60 | +0.55 | +0.05 | +0.21 |
| qwen-plus (rated) | +0.60 | +0.63 | +0.07 | +0.18 |
| gpt-5.3-chat (rated) | +0.50 | +0.67 | +0.09 | +0.16 |
| deepseek-v4-pro (rated) | +0.55 | +0.73 | +0.15 | +0.10 |
| llama-4-scout (rated) | +0.62 | +0.53 | +0.05 | +0.20 |
| qwen3-vl-30b-a3b-instruct (rated) | +0.53 | +0.53 | +0.06 | +0.19 |
| qwen3.5-397b-a17b (rated) | +0.54 | +0.65 | +0.08 | +0.17 |
| qwen3.5-plus-02-15 (rated) | +0.54 | +0.66 | +0.08 | +0.17 |
| glm-5.3 (rated) | +0.55 | +0.64 | +0.12 | +0.13 |
| qwen3.7-plus (rated) | +0.51 | +0.67 | +0.10 | +0.15 |
| qwen3-30b-a3b-instruct-2507 (rated) | +0.65 | +0.54 | +0.05 | +0.20 |
| gpt-5.5 (rated) | +0.42 | +0.76 | +0.17 | +0.07 |
| qwen3.6-35b-a3b (rated) | +0.61 | +0.59 | +0.10 | +0.14 |
| qwen3.6-flash (rated) | +0.65 | +0.58 | +0.09 | +0.15 |
| qwen3.6-plus (rated) | +0.52 | +0.67 | +0.08 | +0.15 |
| qwen3-235b-a22b (rated) | +0.64 | +0.56 | +0.07 | +0.17 |
| qwen-2.5-72b-instruct (rated) | +0.60 | +0.60 | +0.06 | +0.17 |
| qwen-plus-2025-07-28 (rated) | +0.64 | +0.51 | +0.07 | +0.16 |
| qwen3.7-max (rated) | +0.50 | +0.65 | +0.11 | +0.12 |
| qwen3-235b-a22b-2507 (rated) | +0.64 | +0.57 | +0.07 | +0.16 |
| qwen3.5-9b (rated) | +0.50 | +0.59 | +0.09 | +0.14 |
| qwen2.5-vl-72b-instruct (rated) | +0.58 | +0.60 | +0.07 | +0.15 |
| qwen3-vl-32b-instruct (rated) | +0.60 | +0.53 | +0.02 | +0.20 |
| gpt-5.6-sol (rated) | +0.53 | +0.67 | +0.09 | +0.13 |
| qwen3-max-thinking (rated) | +0.58 | +0.70 | +0.07 | +0.14 |
| qwen3-coder-plus (rated) | +0.62 | +0.60 | +0.07 | +0.14 |
| inkling (rated) | +0.56 | +0.69 | +0.10 | +0.11 |
| qwen3-vl-8b-instruct (rated) | +0.61 | +0.60 | +0.06 | +0.15 |
| qwen-2.5-7b-instruct (rated) | +0.55 | +0.59 | +0.05 | +0.16 |
| kimi-k3 (rated) | +0.62 | +0.67 | +0.07 | +0.14 |
| qwen3.5-27b (rated) | +0.60 | +0.65 | +0.05 | +0.16 |
| qwen3-32b (rated) | +0.61 | +0.54 | +0.06 | +0.14 |
| qwen3-14b (rated) | +0.61 | +0.58 | +0.03 | +0.16 |
| qwen3-coder-next (rated) | +0.64 | +0.58 | +0.05 | +0.13 |
| qwen3-next-80b-a3b-instruct (rated) | +0.67 | +0.70 | +0.05 | +0.12 |
| claude-opus-4.7 (rated) | +0.59 | +0.60 | +0.07 | +0.09 |
| qwen3-8b (rated) | +0.62 | +0.54 | +0.03 | +0.12 |
| claude-opus-4.8 (rated) | +0.61 | +0.58 | +0.07 | +0.07 |
| gemini-3.7-flash (rated) | +0.50 | +0.63 | +0.03 | +0.10 |
| claude-fable-5.1 (rated) | +0.58 | +0.61 | +0.07 | +0.05 |
| claude-opus-4.6 (rated) | +0.63 | +0.63 | +0.04 | +0.05 |
| model | x self-expr | y secular | x 95%CI | y 95%CI |
|:--------------------------------------|--------------:|------------:|----------:|----------:|
| gemini-3.1-flash-lite (rated) | +0.56 | +0.57 | +0.14 | +0.19 |
| gemini-3.1-flash-lite-preview (rated) | +0.58 | +0.61 | +0.13 | +0.20 |
| grok-4.3 (rated) | +0.55 | +0.57 | +0.12 | +0.21 |
| qwen3.6-27b (rated) | +0.46 | +0.68 | +0.18 | +0.14 |
| qwen3.5-35b-a3b (rated) | +0.54 | +0.55 | +0.14 | +0.18 |
| qwen3.6-max-preview (rated) | +0.48 | +0.69 | +0.16 | +0.14 |
| muse-spark-1.3 (rated) | +0.43 | +0.73 | +0.18 | +0.12 |
| qwen3.8-27b (rated) | +0.43 | +0.69 | +0.16 | +0.14 |
| deepseek-v4-flash-0731 (rated) | +0.55 | +0.58 | +0.13 | +0.16 |
| gpt-oss-120b (rated) | +0.48 | +0.60 | +0.12 | +0.18 |
| qwen3.8-flash (rated) | +0.42 | +0.63 | +0.14 | +0.15 |
| gemma-4-31b-it (rated) | +0.36 | +0.66 | +0.16 | +0.13 |
| qwen3.5-122b-a10b (rated) | +0.48 | +0.62 | +0.15 | +0.14 |
| glm-4.5-air (rated) | +0.59 | +0.64 | +0.13 | +0.15 |
| gpt-5.4 (rated) | +0.62 | +0.64 | +0.12 | +0.17 |
| gemini-3.5-flash-lite (rated) | +0.57 | +0.65 | +0.13 | +0.15 |
| qwen3.5-plus-20260420 (rated) | +0.59 | +0.65 | +0.12 | +0.16 |
| gpt-5-mini (rated) | +0.52 | +0.64 | +0.09 | +0.19 |
| gpt-5.6-luna (rated) | +0.60 | +0.58 | +0.09 | +0.19 |
| gpt-5.6-terra (rated) | +0.55 | +0.64 | +0.10 | +0.17 |
| grok-4.6 (rated) | +0.42 | +0.72 | +0.15 | +0.13 |
| glm-5.1 (rated) | +0.54 | +0.69 | +0.12 | +0.15 |
| gpt-5.2 (rated) | +0.55 | +0.64 | +0.10 | +0.17 |
| gemini-2.5-pro (rated) | +0.47 | +0.70 | +0.13 | +0.14 |
| gpt-5.4-mini (rated) | +0.46 | +0.67 | +0.11 | +0.16 |
| grok-4.20 (rated) | +0.52 | +0.64 | +0.10 | +0.16 |
| gpt-4o-mini (rated) | +0.61 | +0.57 | +0.11 | +0.16 |
| deepseek-v4.1-flash (rated) | +0.54 | +0.60 | +0.10 | +0.16 |
| qwen3-coder-30b-a3b-instruct (rated) | +0.50 | +0.58 | +0.09 | +0.17 |
| glm-5.2 (rated) | +0.53 | +0.69 | +0.12 | +0.14 |
| deepseek-v3.2-exp (rated) | +0.58 | +0.60 | +0.09 | +0.17 |
| qwen3-30b-a3b (rated) | +0.67 | +0.61 | +0.07 | +0.19 |
| gpt-6-astra (rated) | +0.46 | +0.68 | +0.15 | +0.11 |
| mistral-large-2512 (rated) | +0.65 | +0.64 | +0.08 | +0.18 |
| gemma-3-27b-it (rated) | +0.60 | +0.60 | +0.09 | +0.17 |
| llama-4-maverick (rated) | +0.64 | +0.64 | +0.08 | +0.18 |
| qwen3-coder (rated) | +0.60 | +0.62 | +0.08 | +0.18 |
| qwen3.7-flash (rated) | +0.65 | +0.59 | +0.10 | +0.16 |
| qwen3-vl-235b-a22b-instruct (rated) | +0.60 | +0.55 | +0.05 | +0.21 |
| qwen-plus (rated) | +0.60 | +0.63 | +0.07 | +0.18 |
| gpt-5.3-chat (rated) | +0.50 | +0.67 | +0.09 | +0.16 |
| deepseek-v4-pro (rated) | +0.55 | +0.73 | +0.15 | +0.10 |
| llama-4-scout (rated) | +0.62 | +0.53 | +0.05 | +0.20 |
| grok-4.5 (rated) | +0.52 | +0.71 | +0.12 | +0.13 |
| gpt-4o-mini-2024-07-18 (rated) | +0.63 | +0.57 | +0.09 | +0.16 |
| qwen3-vl-30b-a3b-instruct (rated) | +0.53 | +0.53 | +0.06 | +0.19 |
| qwen3.5-397b-a17b (rated) | +0.54 | +0.65 | +0.08 | +0.17 |
| deepseek-chat-v3.1 (rated) | +0.62 | +0.60 | +0.07 | +0.18 |
| qwen3.5-plus-02-15 (rated) | +0.54 | +0.66 | +0.08 | +0.17 |
| glm-5.3 (rated) | +0.55 | +0.64 | +0.12 | +0.13 |
| deepseek-v3.2 (rated) | +0.57 | +0.56 | +0.08 | +0.17 |
| kimi-k2.6 (rated) | +0.61 | +0.67 | +0.09 | +0.16 |
| glm-5 (rated) | +0.59 | +0.67 | +0.08 | +0.16 |
| qwen3.7-plus (rated) | +0.51 | +0.67 | +0.10 | +0.15 |
| qwen3-30b-a3b-instruct-2507 (rated) | +0.65 | +0.54 | +0.05 | +0.20 |
| gpt-5.5 (rated) | +0.42 | +0.76 | +0.17 | +0.07 |
| qwen3.6-35b-a3b (rated) | +0.61 | +0.59 | +0.10 | +0.14 |
| qwen3.6-flash (rated) | +0.65 | +0.58 | +0.09 | +0.15 |
| gemini-3.8-flash (rated) | +0.38 | +0.64 | +0.16 | +0.08 |
| deepseek-chat-v3-0324 (rated) | +0.59 | +0.60 | +0.07 | +0.17 |
| qwen3.6-plus (rated) | +0.52 | +0.67 | +0.08 | +0.15 |
| qwen3-235b-a22b (rated) | +0.64 | +0.56 | +0.07 | +0.17 |
| gpt-5.4-nano (rated) | +0.48 | +0.60 | +0.10 | +0.13 |
| gemini-3.5-flash (rated) | +0.59 | +0.58 | +0.08 | +0.15 |
| qwen-2.5-72b-instruct (rated) | +0.60 | +0.60 | +0.06 | +0.17 |
| deepseek-v4-flash (rated) | +0.59 | +0.63 | +0.09 | +0.14 |
| qwen-plus-2025-07-28 (rated) | +0.64 | +0.51 | +0.07 | +0.16 |
| gemini-3-flash-preview (rated) | +0.56 | +0.63 | +0.10 | +0.13 |
| qwen3.7-max (rated) | +0.50 | +0.65 | +0.11 | +0.12 |
| qwen3-235b-a22b-2507 (rated) | +0.64 | +0.57 | +0.07 | +0.16 |
| qwen3.5-9b (rated) | +0.50 | +0.59 | +0.09 | +0.14 |
| muse-glimmer-30b (rated) | +0.49 | +0.65 | +0.10 | +0.13 |
| gemini-2.5-flash (rated) | +0.63 | +0.65 | +0.08 | +0.15 |
| qwen2.5-vl-72b-instruct (rated) | +0.58 | +0.60 | +0.07 | +0.15 |
| qwen3-vl-32b-instruct (rated) | +0.60 | +0.53 | +0.02 | +0.20 |
| gemini-2.5-flash-lite (rated) | +0.39 | +0.67 | +0.09 | +0.13 |
| gpt-5.6-sol (rated) | +0.53 | +0.67 | +0.09 | +0.13 |
| qwen3-max-thinking (rated) | +0.58 | +0.70 | +0.07 | +0.14 |
| qwen3-coder-plus (rated) | +0.62 | +0.60 | +0.07 | +0.14 |
| inkling (rated) | +0.56 | +0.69 | +0.10 | +0.11 |
| qwen3-vl-8b-instruct (rated) | +0.61 | +0.60 | +0.06 | +0.15 |
| qwen-2.5-7b-instruct (rated) | +0.55 | +0.59 | +0.05 | +0.16 |
| kimi-k3 (rated) | +0.62 | +0.67 | +0.07 | +0.14 |
| gpt-4o-2024-08-06 (rated) | +0.59 | +0.58 | +0.08 | +0.13 |
| qwen3.5-27b (rated) | +0.60 | +0.65 | +0.05 | +0.16 |
| qwen3.8-2.4t-a95b (rated) | +0.53 | +0.62 | +0.07 | +0.13 |
| qwen3-32b (rated) | +0.61 | +0.54 | +0.06 | +0.14 |
| qwen3-14b (rated) | +0.61 | +0.58 | +0.03 | +0.16 |
| glm-5.3-flash (rated) | +0.54 | +0.64 | +0.09 | +0.11 |
| glm-4.7-flash (rated) | +0.56 | +0.59 | +0.08 | +0.11 |
| gemini-3.6-flash (rated) | +0.59 | +0.62 | +0.07 | +0.12 |
| gpt-4.1 (rated) | +0.63 | +0.69 | +0.06 | +0.14 |
| gpt-4o (rated) | +0.61 | +0.58 | +0.07 | +0.12 |
| gpt-5.1 (rated) | +0.61 | +0.75 | +0.09 | +0.10 |
| qwen3-coder-next (rated) | +0.64 | +0.58 | +0.05 | +0.13 |
| gpt-4.1-mini (rated) | +0.57 | +0.45 | +0.10 | +0.08 |
| gpt-5 (rated) | +0.57 | +0.72 | +0.07 | +0.11 |
| qwen3-next-80b-a3b-instruct (rated) | +0.67 | +0.70 | +0.05 | +0.12 |
| gpt-4o-2024-11-20 (rated) | +0.62 | +0.60 | +0.06 | +0.10 |
| gpt-4.1-nano (rated) | +0.48 | +0.50 | +0.06 | +0.10 |
| gpt-3.5-turbo-0613 (rated) | +0.61 | +0.44 | +0.08 | +0.08 |
| claude-opus-4.7 (rated) | +0.59 | +0.60 | +0.07 | +0.09 |
| gpt-3.5-turbo-16k (rated) | +0.61 | +0.42 | +0.08 | +0.07 |
| qwen3-8b (rated) | +0.62 | +0.54 | +0.03 | +0.12 |
| claude-opus-4.8 (rated) | +0.61 | +0.58 | +0.07 | +0.07 |
| gemini-3.7-flash (rated) | +0.50 | +0.63 | +0.03 | +0.10 |
| claude-fable-5.1 (rated) | +0.58 | +0.61 | +0.07 | +0.05 |
| claude-opus-4.6 (rated) | +0.63 | +0.63 | +0.04 | +0.05 |
+1 -1
View File
@@ -4,7 +4,7 @@
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Moral Maps: Where Do Frontier Models' Cultural Values Lie?</title>
<script type="module" crossorigin src="./assets/index-BZu_5V-r.js"></script>
<script type="module" crossorigin src="./assets/index-DVN-FNIK.js"></script>
<link rel="stylesheet" crossorigin href="./assets/index-CiBO8OQd.css">
</head>
<body>
File diff suppressed because it is too large Load Diff
+8 -7
View File
@@ -553,15 +553,15 @@ def main() -> None:
saved_catalog[panel["model"]]["created"], tz=timezone.utc).date().isoformat()}
if panel is None:
legacy_date = legacy_release_dates.get(name)
provenance = {"readout": "recovered rounded historical coordinate", "items": None,
"samples": None, "run_id": None, "protocol_id": None,
provenance = {"coordinate_provenance": "historical rounded coordinate", "readout": "recovered rounded historical coordinate",
"eval_version": None, "items": None, "samples": None, "run_id": None, "protocol_id": None,
"release_created": legacy_date,
"release_source": (f"saved OpenRouter catalog 2026-09-17: {LEGACY_CATALOG_IDS[name]}"
if legacy_date else "historical coordinate")}
else:
provenance = {"readout": "rated categorical response", "items": panel["n_items"],
"samples": panel["n_samples"], "run_id": panel["run_id"],
"protocol_id": panel["protocol_id"],
provenance = {"coordinate_provenance": "canonical score-all-options", "readout": "rated categorical response",
"eval_version": panel["eval_version"], "items": panel["n_items"], "samples": panel["n_samples"],
"run_id": panel["run_id"], "protocol_id": panel["protocol_id"],
"release_created": catalog["created"] if catalog else None,
"release_source": "catalog" if catalog else "request ledger"}
capability = capability_by_model.get(name)
@@ -612,8 +612,9 @@ def main() -> None:
"WVS Inglehart-Welzel", countries, P,
("Survival", "Self-expression", "Traditional", "Secular-Rational"),
models=plot_models, model_labels=model_labels, emphasize=emph,
title="Frontier LLMs on the\nWorld Values Survey",
note=f"{len(plot_models)} models, rated sampling\ngithub.com/wassname/moral-maps")
title="Moral Maps: Where Do Frontier\nModels' Cultural Values Lie?",
note="World Values Survey | source: github.com/wassname/moral-maps",
title_y=0.115, note_y=0.04)
fig.savefig(args.out, dpi=200, bbox_inches="tight")
fig.savefig(Path(args.out).with_suffix(".svg"), bbox_inches="tight")
logger.info(f"wrote {args.out}")
+5 -2
View File
@@ -88,7 +88,10 @@ function ReleaseScatter({ data, hidden, field, title, axisMode }) {
}, [axisMode, visible]);
const frontierPlacement = useMemo(() => {
const bounds = { left, right: width - right, top, bottom: height - bottom };
const taken = plotted.map(model => box(plotX(xValue(model)), valueY(coordinate(model)), 16, 16));
const taken = [
...plotted.map(model => box(plotX(xValue(model)), valueY(coordinate(model)), 16, 16)),
...(fit ? [box(left + 66, top + 9, 128, 18)] : []),
];
const placement = {};
for (const model of frontier) {
const anchor = { x: plotX(xValue(model)), y: valueY(coordinate(model)) };
@@ -213,7 +216,7 @@ function Map({ data }) {
{data.zone_hulls.map(zone => <text key={zone.name} className="zone-label" x={labels[`zone:${zone.name}`].cx} y={labels[`zone:${zone.name}`].cy + 5} textAnchor="middle" fill={zone.color}>{zone.name}</text>)}
{Object.entries(groups).map(([family, models]) => <g key={family} data-family={family} display={hidden.has(family) ? 'none' : 'inline'}>{models.map(model => <ModelMarker key={model.name} model={model} placement={labels} geometry={geometry} setActive={setActive} clearActive={clearActive} markerRef={model.name === focusName ? focusRef : null} logo={data.logos[model.family]} />)}</g>)}
<g className="poles"><line x1={xMedian} y1="62" x2={xMedian} y2={geometry.bounds.top} markerEnd="url(#arrow)" /><line x1={xMedian} y1={geometry.bounds.bottom} x2={xMedian} y2="838" markerEnd="url(#arrow)" /><line x1="64" y1={yMedian} x2={geometry.bounds.left} y2={yMedian} markerEnd="url(#arrow)" /><line x1={geometry.bounds.right} y1={yMedian} x2="1184" y2={yMedian} markerEnd="url(#arrow)" /><text x={xMedian} y="40" textAnchor="middle">{data.axis.y[1]}</text><text x={xMedian} y="870" textAnchor="middle">{data.axis.y[0]}</text><text x="25" y={yMedian + 7}>{data.axis.x[0]}</text><text x="1136" y={yMedian + 7} textAnchor="end">{data.axis.x[1]}</text></g>
<text className="map-title" x={geometry.bounds.left + 8} y={geometry.bounds.bottom - 54}>{data.title.split('\n').map((line, index) => <tspan key={line} x={geometry.bounds.left + 8} dy={index ? 17 : 0}>{line}</tspan>)}</text>
<text className="map-title" x={geometry.bounds.left + 8} y={geometry.bounds.bottom - 78}>{data.title.split('\n').map((line, index) => <tspan key={line} x={geometry.bounds.left + 8} dy={index ? 17 : 0}>{line}</tspan>)}</text>
<text className="map-note" x={geometry.bounds.left + 8} y={geometry.bounds.bottom - 34} textAnchor="start">{data.note.split('\n').map((line, index) => <tspan key={line} x={geometry.bounds.left + 8} dy={index ? 11 : 0}>{line}</tspan>)}</text>
</svg>
<Tooltip active={active} geometry={geometry} />
+39 -15
View File
@@ -57,15 +57,27 @@ def main() -> None:
assert data["capability_x"]["fetched_utc"] == "2026-09-17T10:13:20Z"
capability_matched = [model for model in data["models"] if model["provenance"]["hle_score"] is not None]
assert len(capability_matched) == data["capability_x"]["matched_models"] == 46
canonical_models = [model for model in data["models"]
if model["provenance"]["coordinate_provenance"] == "canonical score-all-options"]
historical_models = [model for model in data["models"]
if model["provenance"]["coordinate_provenance"] == "historical rounded coordinate"]
assert len(canonical_models) == 96
assert len(historical_models) == 12
assert all(model["provenance"]["eval_version"] == "wvs-score-all-options-v1"
for model in canonical_models)
assert all(model["provenance"]["eval_version"] is None for model in historical_models)
assert all("reliability" not in str(model["provenance"]["protocol_id"])
for model in canonical_models)
assert len(data["countries"]) == 90
assert len(data["zone_hulls"]) == 4
assert len(data["latest_by_family"]) == 13
assert all(len(zone["points"]) >= 30 for zone in data["zone_hulls"])
dated = [model for model in data["models"] if model["provenance"]["release_created"]]
grok_dates = {model["name"]: model["provenance"]["release_created"] for model in data["models"] if model["name"] in {"grok-4.20", "grok-4.3"}}
assert grok_dates == {"grok-4.20": "2026-03-31", "grok-4.3": "2026-04-30"}
assert all(model["provenance"]["release_source"].startswith("saved OpenRouter catalog 2026-09-17: x-ai/")
for model in data["models"] if model["name"] in grok_dates)
dated = sorted((model for model in data["models"] if model["provenance"]["release_created"]),
key=lambda model: (model["provenance"]["release_created"], model["name"]))
grok_dates = {model["name"]: model["provenance"]["release_created"] for model in data["models"] if model["family"] == "grok"}
assert grok_dates == {"grok-4.20": "2026-03-31", "grok-4.3": "2026-04-30",
"grok-4.5": "2026-07-08", "grok-4.6": "2026-08-12"}
assert data["latest_by_family"]["grok"]["name"] == "grok-4.6"
families = {model["family"] for model in data["models"]}
with sync_playwright() as playwright:
@@ -130,9 +142,20 @@ def main() -> None:
svg_description(page, '.release-panel[data-coordinate="x"] svg', "release-x-svg-title", "release-x-svg-desc")
assert page.locator(".release-mark").count() == len(dated) * 2
assert page.locator(".frontier-label").count() > 0
frontier_boxes = page.locator('.frontier-label text').evaluate_all("nodes => nodes.map(node => { const b = node.getBBox(); return [b.x, b.y, b.width, b.height]; })")
assert all(x >= 96 and y >= 42 and x + width <= 1165 and y + height <= 272 for x, y, width, height in frontier_boxes)
assert all(a[0] + a[2] <= b[0] or b[0] + b[2] <= a[0] or a[1] + a[3] <= b[1] or b[1] + b[3] <= a[1] for index, a in enumerate(frontier_boxes) for b in frontier_boxes[index + 1:])
expected_frontier = []
best_score = float("-inf")
for model in dated:
score = model["provenance"]["hle_score"]
if score is not None and score > best_score:
expected_frontier.append(model["name"])
best_score = score
for panel in page.locator(".release-panel").all():
assert panel.locator(".frontier-label line").count() == 0
actual_frontier = panel.locator(".frontier-label").evaluate_all("nodes => nodes.map(node => node.dataset.frontierModel)")
assert actual_frontier == expected_frontier, (actual_frontier, expected_frontier)
frontier_boxes = panel.locator('.frontier-label text').evaluate_all("nodes => nodes.map(node => { const b = node.getBBox(); return [b.x, b.y, b.width, b.height]; })")
assert all(x >= 96 and y >= 42 and x + width <= 1165 and y + height <= 272 for x, y, width, height in frontier_boxes)
assert all(a[0] + a[2] <= b[0] or b[0] + b[2] <= a[0] or a[1] + a[3] <= b[1] or b[1] + b[3] <= a[1] for index, a in enumerate(frontier_boxes) for b in frontier_boxes[index + 1:])
for grok_name in grok_dates:
assert page.locator(f'[data-release-model="{grok_name}"]').count() == 2
self_expression_panel = page.locator('.release-panel[data-coordinate="x"]')
@@ -198,7 +221,7 @@ def main() -> None:
page.screenshot(path=OUT / "wvs_react_root_playwright_keyboard_focus.png", full_page=True)
release_y_marker = page.locator('.release-panel[data-coordinate="y"] [data-release-model="qwen3.8-flash"]')
release_y_marker.locator(".model-ring").hover()
release_y_marker.focus()
page.wait_for_selector("#release-y-tooltip")
assert page.locator("#release-y-tooltip strong").inner_text() == "qwen3.8-flash"
assert page.locator("#release-y-tooltip span").all_text_contents() == ["Secular-Rational: 0.635", "release 2026-08-26"]
@@ -208,7 +231,7 @@ def main() -> None:
assert page.locator("#release-y-tooltip").is_visible()
page.screenshot(path=OUT / "wvs_react_root_playwright_release_panel_keyboard_focus.png", full_page=True)
release_x_marker = page.locator('.release-panel[data-coordinate="x"] [data-release-model="qwen3.8-flash"]')
release_x_marker.locator(".model-ring").hover()
release_x_marker.focus()
page.wait_for_selector("#release-x-tooltip")
assert release_y_marker.get_attribute("aria-describedby") == "release-y-tooltip"
assert release_x_marker.get_attribute("aria-describedby") == "release-x-tooltip"
@@ -218,7 +241,7 @@ def main() -> None:
for grok_name, grok_date in grok_dates.items():
grok = next(model for model in dated if model["name"] == grok_name)
grok_marker = page.locator(f'.release-panel[data-coordinate="y"] [data-release-model="{grok_name}"]')
grok_marker.locator(".model-ring").hover()
grok_marker.focus()
assert page.locator("#release-y-tooltip strong").inner_text() == grok_name
assert page.locator("#release-y-tooltip span").all_text_contents() == [
f"Secular-Rational: {grok['y']:.3f}", f"release {grok_date}",
@@ -238,10 +261,10 @@ def main() -> None:
page.screenshot(path=OUT / "wvs_react_playwright_release_fit_qwen_only.png", full_page=True)
page.get_by_role("button", name="qwen", exact=True).click()
page.get_by_role("button", name="muse", exact=True).click()
page.get_by_role("button", name="inkling", exact=True).click()
page.wait_for_timeout(100)
for family in families:
expected = "inline" if family == "muse" else "none"
expected = "inline" if family == "inkling" else "none"
assert map_svg.locator(f':scope > g[data-family="{family}"]').get_attribute("display") == expected
for panel in page.locator(".release-panel").all():
assert int(panel.locator("svg").get_attribute("data-fit-n")) == 0
@@ -268,12 +291,13 @@ def main() -> None:
print(json.dumps({
"root": "React app loaded at /",
"compatibility": "/wvs/ and /wvs/react/ redirect to the root React app; shared data remains at /wvs/wvs_map_data.json",
"shared": {"models": 64, "countries": 90, "buffered_hulls": 4, "latest_labels": 13,
"shared": {"models": len(data["models"]), "canonical_score_all_options": len(canonical_models),
"historical_rounded": len(historical_models), "countries": 90, "buffered_hulls": 4, "latest_labels": 13,
"numeric_model_country_median_equality": "verified against DOM data attributes"},
"svg_accessibility": "main map and both release panels have stable title/desc aria-labelledby; descriptions update after family visibility change",
"qwen_toggle": "clicked, map and dated-panel family groups hidden, country coordinates invariant",
"tooltip": "map and release-panel pointer hover and focus show panel-specific model name, coordinate, and release date only",
"release_panels": {"panels": 2, "dated_models": len(dated), "date_order": "DOM order verified; both catalog-dated Grok models rendered in each panel",
"release_panels": {"panels": 2, "dated_models": len(dated), "date_order": "DOM order verified; all four catalog-dated Grok models rendered in each panel",
"ols": "line, n, and R squared recompute from currently visible matched models; the Self-expression panel renders negative stored x, so upward means Self-expression; fewer than two distinct x values hide the fit"},
"capability_panels": {"matched_models": len(capability_matched), "omitted_models": len(data["models"]) - len(capability_matched),
"source": data["capability_x"]["source_url"], "fetched_utc": data["capability_x"]["fetched_utc"],
@@ -13,7 +13,7 @@ from pathlib import Path
import numpy as np
from moralmaps.iw_axes import X_AXIS, Y_AXIS, resolve_items
from moralmaps.rated_cache import merge_completed, update_coords
from moralmaps.rated_cache import merge_completed, update_coords, update_eval_versions
from wvs_map import load_wvs_all, model_coord_ci
CACHE = Path("slop/research/wvs/20260916_openrouter/wvs_iw_rated.json")
@@ -101,6 +101,8 @@ def main() -> None:
parser.add_argument("--smoke", action="store_true")
parser.add_argument("--refresh-ci", action="store_true",
help="recompute only existing complete panels' coordinate CI summaries from the ledger")
parser.add_argument("--stamp-eval-version", action="store_true",
help="add score-all-options v1 provenance to cache entries recovered from the ledger")
args = parser.parse_args()
if args.smoke:
concurrency_smoke()
@@ -120,6 +122,8 @@ def main() -> None:
merged = update_coords(CACHE, {key: entry["coords"] for key, entry in overlaps.items()})
else:
merged = merge_completed(CACHE, additions)
if args.stamp_eval_version:
merged = update_eval_versions(CACHE, {key: "wvs-score-all-options-v1" for key in entries})
preserved = {key: merged["completed"][key] for key in existing}
preserved_hash = hashlib.sha256(json.dumps({key: stable_entry(value) for key, value in preserved.items()},
sort_keys=True, separators=(",", ":")).encode()).hexdigest()
@@ -131,6 +135,7 @@ def main() -> None:
"existing_entries_sha256_after": preserved_hash,
"new_complete_runs_added": len(additions),
"ci_summaries_refreshed": len(overlaps) if args.refresh_ci else 0,
"eval_versions_stamped": len(entries) if args.stamp_eval_version else 0,
"new_protocol_ids": sorted(additions),
"new_models": sorted(entry["model"] for entry in additions.values()),
"overlap_complete_runs": len(overlaps),
@@ -1,14 +1,15 @@
{
"cache_completed_entries_after_merge": 86,
"ci_summaries_refreshed": 86,
"existing_entries_preserved_count": 86,
"existing_entries_sha256_after": "2fa381c2744633f337988c3fc2a753227849a733aaa906eceb507fe60b5b187f",
"existing_entries_sha256_before": "2fa381c2744633f337988c3fc2a753227849a733aaa906eceb507fe60b5b187f",
"cache_completed_entries_after_merge": 97,
"ci_summaries_refreshed": 0,
"eval_versions_stamped": 97,
"existing_entries_preserved_count": 97,
"existing_entries_sha256_after": "b4b6afcfa52ca47008d26ab7ac923fd741c0626690e02298cafb7e55f3c6dcd3",
"existing_entries_sha256_before": "47de431087eeea68bbf2591a8152fcb5c1fa6a084cace7fa5a6183ee635c7f3f",
"ledger": "slop/research/wvs/20260916_openrouter/wvs_iw_requests.jsonl",
"ledger_valid_lines": 49466,
"ledger_valid_lines": 54700,
"new_complete_runs_added": 0,
"new_models": [],
"new_protocol_ids": [],
"overlap_complete_runs": 86,
"overlap_complete_runs": 97,
"overlap_point_coordinates_equal": true
}
@@ -0,0 +1,22 @@
# Final canonical WVS integration
-- PI[k3]
## Observations
- `wvs_iw_rated.json` has 97 complete cache records. `gpt-5-nano` remains excluded from the map because its prior content-quality audit documented systematic neutral/example-like outputs. The public artifact has 108 distinct names: 96 canonical `wvs-score-all-options-v1` entries and 12 distinct historical rounded coordinates. Four historical names overlap canonical names, so the canonical coordinate is shown for each overlap.
- The public artifact contains 13 families. The Grok entries are `grok-4.20`, `grok-4.3`, `grok-4.5`, and `grok-4.6`; the catalog-date latest label is `grok-4.6` (2026-08-12).
- The saved Artificial Analysis HLE mapping supplies 46 matches and 62 omissions among the 108 displayed names. The artifact retains the source URL and fetch time, `2026-09-17T10:13:20Z`.
- No displayed provenance protocol ID belongs to the DeepSeek reliability pilot. Reliability aggregate and incomplete replicate records were not used to generate the map.
## Verification
`uv run --no-project --offline --with 'playwright==1.62.0' python scripts/wvs_react/uat.py` passed. It checks the root and both redirects, 13 family controls, map/country geometry, all four Groks, `grok-4.6` as latest, canonical versus historical provenance, absent pilot IDs, release-date frontier labels without leader lines, capability fits under all-family and family-subset visibility, and accessible SVG titles/descriptions.
I inspected `slop/research/wvs/20260916_openrouter/wvs_react_root_playwright_default.png`. A fresh-eyes review also found the lower-left title/source readable and the revised OLS/frontier labels separated. Its evidence is a read-only reviewer response in this worker session, not a committed file.
## Exclusions
- `gpt-5-nano` is a completed cache diagnostic but is not plotted because its existing content-quality audit excluded it.
- Incomplete/partial score-all-options panels and all DeepSeek reliability-pilot data are excluded.
- The 16 historical rounded inputs remain distinct provenance; four duplicate canonical names resolve to the canonical panel.
File diff suppressed because it is too large Load Diff
+4 -3
View File
@@ -298,7 +298,8 @@ def plot_value_map(display: str, countries: list[str], P: np.ndarray,
model_labels: dict[str, str] | None = None,
steer: dict[str, tuple[float, float, str]] | None = None,
emphasize: set[str] | None = None,
title: str | None = None, note: str | None = None):
title: str | None = None, note: str | None = None,
title_y: float = 0.075, note_y: float = 0.02):
"""The interpretable "4-value map": two NAMED axes with four pole signposts through the human
MEDIAN crosshair, Economist-style zone hulls (the 4 most-separate zones), zone-coloured dots, and
auto-placed labels (landmarks + corner outliers + one representative per zone + any models; see
@@ -420,10 +421,10 @@ def plot_value_map(display: str, countries: list[str], P: np.ndarray,
# INSIDE the axes, in the empty bottom-left corner, so a crop of the PNG can't strip the attribution
# and there's no external white band. -- Claude
if title:
ax.text(0.012, 0.075, title, transform=ax.transAxes, ha="left", va="bottom",
ax.text(0.012, title_y, title, transform=ax.transAxes, ha="left", va="bottom",
fontsize=11, fontweight="bold", color="#333", zorder=11, linespacing=1.05)
if note:
ax.text(0.012, 0.02, note, transform=ax.transAxes, ha="left", va="bottom",
ax.text(0.012, note_y, note, transform=ax.transAxes, ha="left", va="bottom",
fontsize=7.5, color="#8a857a", zorder=11, linespacing=1.1)
return fig
+8
View File
@@ -37,3 +37,11 @@ def update_coords(path: Path, coords: dict[str, list[float]]) -> dict:
completed[protocol_id]["coords"] = values
completed[protocol_id]["ci_method"] = "combined item and N-response-mean bootstrap"
return _write_locked(path, update)
def update_eval_versions(path: Path, versions: dict[str, str]) -> dict:
"""Add evaluator provenance without changing cached coordinates or run records."""
def update(completed: dict[str, dict]) -> None:
for protocol_id, version in versions.items():
completed[protocol_id]["eval_version"] = version
return _write_locked(path, update)