mirror of
https://github.com/wassname/moral-maps.git
synced 2026-09-24 13:41:05 +08:00
Refresh canonical WVS map integration
Co-Authored-By: PI[k3] <288921227+claudypoo@users.noreply.github.com>
This commit is contained in:
File diff suppressed because one or more lines are too long
Binary file not shown.
|
Before Width: | Height: | Size: 404 KiB After Width: | Height: | Size: 444 KiB |
+4980
-4445
File diff suppressed because it is too large
Load Diff
|
Before Width: | Height: | Size: 493 KiB After Width: | Height: | Size: 515 KiB |
+110
-66
@@ -1,66 +1,110 @@
|
||||
| model | x self-expr | y secular | x 95%CI | y 95%CI |
|
||||
|:-------------------------------------|--------------:|------------:|----------:|----------:|
|
||||
| qwen3.6-27b (rated) | +0.46 | +0.68 | +0.18 | +0.14 |
|
||||
| qwen3.5-35b-a3b (rated) | +0.54 | +0.55 | +0.14 | +0.18 |
|
||||
| qwen3.6-max-preview (rated) | +0.48 | +0.69 | +0.16 | +0.14 |
|
||||
| muse-spark-1.3 (rated) | +0.43 | +0.73 | +0.18 | +0.12 |
|
||||
| qwen3.8-27b (rated) | +0.43 | +0.69 | +0.16 | +0.14 |
|
||||
| qwen3.8-flash (rated) | +0.42 | +0.63 | +0.14 | +0.15 |
|
||||
| gemma-4-31b-it (rated) | +0.36 | +0.66 | +0.16 | +0.13 |
|
||||
| qwen3.5-122b-a10b (rated) | +0.48 | +0.62 | +0.15 | +0.14 |
|
||||
| grok-4.20 (rated) | +0.59 | +0.62 | +0.11 | +0.17 |
|
||||
| qwen3.5-plus-20260420 (rated) | +0.59 | +0.65 | +0.12 | +0.16 |
|
||||
| gemini-2.5-pro (rated) | +0.47 | +0.70 | +0.13 | +0.14 |
|
||||
| grok-4.3 (rated) | +0.44 | +0.73 | +0.14 | +0.13 |
|
||||
| deepseek-v4-flash (rated) | +0.57 | +0.64 | +0.11 | +0.16 |
|
||||
| gpt-5.4 (rated) | +0.45 | +0.68 | +0.13 | +0.14 |
|
||||
| deepseek-v4.1-flash (rated) | +0.54 | +0.60 | +0.10 | +0.16 |
|
||||
| qwen3-coder-30b-a3b-instruct (rated) | +0.50 | +0.58 | +0.09 | +0.17 |
|
||||
| qwen3-30b-a3b (rated) | +0.67 | +0.61 | +0.07 | +0.19 |
|
||||
| gpt-6-astra (rated) | +0.46 | +0.68 | +0.15 | +0.11 |
|
||||
| mistral-large-2512 (rated) | +0.65 | +0.64 | +0.08 | +0.18 |
|
||||
| gemma-3-27b-it (rated) | +0.60 | +0.60 | +0.09 | +0.17 |
|
||||
| llama-4-maverick (rated) | +0.64 | +0.64 | +0.08 | +0.18 |
|
||||
| qwen3-coder (rated) | +0.60 | +0.62 | +0.08 | +0.18 |
|
||||
| qwen3.7-flash (rated) | +0.65 | +0.59 | +0.10 | +0.16 |
|
||||
| qwen3-vl-235b-a22b-instruct (rated) | +0.60 | +0.55 | +0.05 | +0.21 |
|
||||
| qwen-plus (rated) | +0.60 | +0.63 | +0.07 | +0.18 |
|
||||
| gpt-5.3-chat (rated) | +0.50 | +0.67 | +0.09 | +0.16 |
|
||||
| deepseek-v4-pro (rated) | +0.55 | +0.73 | +0.15 | +0.10 |
|
||||
| llama-4-scout (rated) | +0.62 | +0.53 | +0.05 | +0.20 |
|
||||
| qwen3-vl-30b-a3b-instruct (rated) | +0.53 | +0.53 | +0.06 | +0.19 |
|
||||
| qwen3.5-397b-a17b (rated) | +0.54 | +0.65 | +0.08 | +0.17 |
|
||||
| qwen3.5-plus-02-15 (rated) | +0.54 | +0.66 | +0.08 | +0.17 |
|
||||
| glm-5.3 (rated) | +0.55 | +0.64 | +0.12 | +0.13 |
|
||||
| qwen3.7-plus (rated) | +0.51 | +0.67 | +0.10 | +0.15 |
|
||||
| qwen3-30b-a3b-instruct-2507 (rated) | +0.65 | +0.54 | +0.05 | +0.20 |
|
||||
| gpt-5.5 (rated) | +0.42 | +0.76 | +0.17 | +0.07 |
|
||||
| qwen3.6-35b-a3b (rated) | +0.61 | +0.59 | +0.10 | +0.14 |
|
||||
| qwen3.6-flash (rated) | +0.65 | +0.58 | +0.09 | +0.15 |
|
||||
| qwen3.6-plus (rated) | +0.52 | +0.67 | +0.08 | +0.15 |
|
||||
| qwen3-235b-a22b (rated) | +0.64 | +0.56 | +0.07 | +0.17 |
|
||||
| qwen-2.5-72b-instruct (rated) | +0.60 | +0.60 | +0.06 | +0.17 |
|
||||
| qwen-plus-2025-07-28 (rated) | +0.64 | +0.51 | +0.07 | +0.16 |
|
||||
| qwen3.7-max (rated) | +0.50 | +0.65 | +0.11 | +0.12 |
|
||||
| qwen3-235b-a22b-2507 (rated) | +0.64 | +0.57 | +0.07 | +0.16 |
|
||||
| qwen3.5-9b (rated) | +0.50 | +0.59 | +0.09 | +0.14 |
|
||||
| qwen2.5-vl-72b-instruct (rated) | +0.58 | +0.60 | +0.07 | +0.15 |
|
||||
| qwen3-vl-32b-instruct (rated) | +0.60 | +0.53 | +0.02 | +0.20 |
|
||||
| gpt-5.6-sol (rated) | +0.53 | +0.67 | +0.09 | +0.13 |
|
||||
| qwen3-max-thinking (rated) | +0.58 | +0.70 | +0.07 | +0.14 |
|
||||
| qwen3-coder-plus (rated) | +0.62 | +0.60 | +0.07 | +0.14 |
|
||||
| inkling (rated) | +0.56 | +0.69 | +0.10 | +0.11 |
|
||||
| qwen3-vl-8b-instruct (rated) | +0.61 | +0.60 | +0.06 | +0.15 |
|
||||
| qwen-2.5-7b-instruct (rated) | +0.55 | +0.59 | +0.05 | +0.16 |
|
||||
| kimi-k3 (rated) | +0.62 | +0.67 | +0.07 | +0.14 |
|
||||
| qwen3.5-27b (rated) | +0.60 | +0.65 | +0.05 | +0.16 |
|
||||
| qwen3-32b (rated) | +0.61 | +0.54 | +0.06 | +0.14 |
|
||||
| qwen3-14b (rated) | +0.61 | +0.58 | +0.03 | +0.16 |
|
||||
| qwen3-coder-next (rated) | +0.64 | +0.58 | +0.05 | +0.13 |
|
||||
| qwen3-next-80b-a3b-instruct (rated) | +0.67 | +0.70 | +0.05 | +0.12 |
|
||||
| claude-opus-4.7 (rated) | +0.59 | +0.60 | +0.07 | +0.09 |
|
||||
| qwen3-8b (rated) | +0.62 | +0.54 | +0.03 | +0.12 |
|
||||
| claude-opus-4.8 (rated) | +0.61 | +0.58 | +0.07 | +0.07 |
|
||||
| gemini-3.7-flash (rated) | +0.50 | +0.63 | +0.03 | +0.10 |
|
||||
| claude-fable-5.1 (rated) | +0.58 | +0.61 | +0.07 | +0.05 |
|
||||
| claude-opus-4.6 (rated) | +0.63 | +0.63 | +0.04 | +0.05 |
|
||||
| model | x self-expr | y secular | x 95%CI | y 95%CI |
|
||||
|:--------------------------------------|--------------:|------------:|----------:|----------:|
|
||||
| gemini-3.1-flash-lite (rated) | +0.56 | +0.57 | +0.14 | +0.19 |
|
||||
| gemini-3.1-flash-lite-preview (rated) | +0.58 | +0.61 | +0.13 | +0.20 |
|
||||
| grok-4.3 (rated) | +0.55 | +0.57 | +0.12 | +0.21 |
|
||||
| qwen3.6-27b (rated) | +0.46 | +0.68 | +0.18 | +0.14 |
|
||||
| qwen3.5-35b-a3b (rated) | +0.54 | +0.55 | +0.14 | +0.18 |
|
||||
| qwen3.6-max-preview (rated) | +0.48 | +0.69 | +0.16 | +0.14 |
|
||||
| muse-spark-1.3 (rated) | +0.43 | +0.73 | +0.18 | +0.12 |
|
||||
| qwen3.8-27b (rated) | +0.43 | +0.69 | +0.16 | +0.14 |
|
||||
| deepseek-v4-flash-0731 (rated) | +0.55 | +0.58 | +0.13 | +0.16 |
|
||||
| gpt-oss-120b (rated) | +0.48 | +0.60 | +0.12 | +0.18 |
|
||||
| qwen3.8-flash (rated) | +0.42 | +0.63 | +0.14 | +0.15 |
|
||||
| gemma-4-31b-it (rated) | +0.36 | +0.66 | +0.16 | +0.13 |
|
||||
| qwen3.5-122b-a10b (rated) | +0.48 | +0.62 | +0.15 | +0.14 |
|
||||
| glm-4.5-air (rated) | +0.59 | +0.64 | +0.13 | +0.15 |
|
||||
| gpt-5.4 (rated) | +0.62 | +0.64 | +0.12 | +0.17 |
|
||||
| gemini-3.5-flash-lite (rated) | +0.57 | +0.65 | +0.13 | +0.15 |
|
||||
| qwen3.5-plus-20260420 (rated) | +0.59 | +0.65 | +0.12 | +0.16 |
|
||||
| gpt-5-mini (rated) | +0.52 | +0.64 | +0.09 | +0.19 |
|
||||
| gpt-5.6-luna (rated) | +0.60 | +0.58 | +0.09 | +0.19 |
|
||||
| gpt-5.6-terra (rated) | +0.55 | +0.64 | +0.10 | +0.17 |
|
||||
| grok-4.6 (rated) | +0.42 | +0.72 | +0.15 | +0.13 |
|
||||
| glm-5.1 (rated) | +0.54 | +0.69 | +0.12 | +0.15 |
|
||||
| gpt-5.2 (rated) | +0.55 | +0.64 | +0.10 | +0.17 |
|
||||
| gemini-2.5-pro (rated) | +0.47 | +0.70 | +0.13 | +0.14 |
|
||||
| gpt-5.4-mini (rated) | +0.46 | +0.67 | +0.11 | +0.16 |
|
||||
| grok-4.20 (rated) | +0.52 | +0.64 | +0.10 | +0.16 |
|
||||
| gpt-4o-mini (rated) | +0.61 | +0.57 | +0.11 | +0.16 |
|
||||
| deepseek-v4.1-flash (rated) | +0.54 | +0.60 | +0.10 | +0.16 |
|
||||
| qwen3-coder-30b-a3b-instruct (rated) | +0.50 | +0.58 | +0.09 | +0.17 |
|
||||
| glm-5.2 (rated) | +0.53 | +0.69 | +0.12 | +0.14 |
|
||||
| deepseek-v3.2-exp (rated) | +0.58 | +0.60 | +0.09 | +0.17 |
|
||||
| qwen3-30b-a3b (rated) | +0.67 | +0.61 | +0.07 | +0.19 |
|
||||
| gpt-6-astra (rated) | +0.46 | +0.68 | +0.15 | +0.11 |
|
||||
| mistral-large-2512 (rated) | +0.65 | +0.64 | +0.08 | +0.18 |
|
||||
| gemma-3-27b-it (rated) | +0.60 | +0.60 | +0.09 | +0.17 |
|
||||
| llama-4-maverick (rated) | +0.64 | +0.64 | +0.08 | +0.18 |
|
||||
| qwen3-coder (rated) | +0.60 | +0.62 | +0.08 | +0.18 |
|
||||
| qwen3.7-flash (rated) | +0.65 | +0.59 | +0.10 | +0.16 |
|
||||
| qwen3-vl-235b-a22b-instruct (rated) | +0.60 | +0.55 | +0.05 | +0.21 |
|
||||
| qwen-plus (rated) | +0.60 | +0.63 | +0.07 | +0.18 |
|
||||
| gpt-5.3-chat (rated) | +0.50 | +0.67 | +0.09 | +0.16 |
|
||||
| deepseek-v4-pro (rated) | +0.55 | +0.73 | +0.15 | +0.10 |
|
||||
| llama-4-scout (rated) | +0.62 | +0.53 | +0.05 | +0.20 |
|
||||
| grok-4.5 (rated) | +0.52 | +0.71 | +0.12 | +0.13 |
|
||||
| gpt-4o-mini-2024-07-18 (rated) | +0.63 | +0.57 | +0.09 | +0.16 |
|
||||
| qwen3-vl-30b-a3b-instruct (rated) | +0.53 | +0.53 | +0.06 | +0.19 |
|
||||
| qwen3.5-397b-a17b (rated) | +0.54 | +0.65 | +0.08 | +0.17 |
|
||||
| deepseek-chat-v3.1 (rated) | +0.62 | +0.60 | +0.07 | +0.18 |
|
||||
| qwen3.5-plus-02-15 (rated) | +0.54 | +0.66 | +0.08 | +0.17 |
|
||||
| glm-5.3 (rated) | +0.55 | +0.64 | +0.12 | +0.13 |
|
||||
| deepseek-v3.2 (rated) | +0.57 | +0.56 | +0.08 | +0.17 |
|
||||
| kimi-k2.6 (rated) | +0.61 | +0.67 | +0.09 | +0.16 |
|
||||
| glm-5 (rated) | +0.59 | +0.67 | +0.08 | +0.16 |
|
||||
| qwen3.7-plus (rated) | +0.51 | +0.67 | +0.10 | +0.15 |
|
||||
| qwen3-30b-a3b-instruct-2507 (rated) | +0.65 | +0.54 | +0.05 | +0.20 |
|
||||
| gpt-5.5 (rated) | +0.42 | +0.76 | +0.17 | +0.07 |
|
||||
| qwen3.6-35b-a3b (rated) | +0.61 | +0.59 | +0.10 | +0.14 |
|
||||
| qwen3.6-flash (rated) | +0.65 | +0.58 | +0.09 | +0.15 |
|
||||
| gemini-3.8-flash (rated) | +0.38 | +0.64 | +0.16 | +0.08 |
|
||||
| deepseek-chat-v3-0324 (rated) | +0.59 | +0.60 | +0.07 | +0.17 |
|
||||
| qwen3.6-plus (rated) | +0.52 | +0.67 | +0.08 | +0.15 |
|
||||
| qwen3-235b-a22b (rated) | +0.64 | +0.56 | +0.07 | +0.17 |
|
||||
| gpt-5.4-nano (rated) | +0.48 | +0.60 | +0.10 | +0.13 |
|
||||
| gemini-3.5-flash (rated) | +0.59 | +0.58 | +0.08 | +0.15 |
|
||||
| qwen-2.5-72b-instruct (rated) | +0.60 | +0.60 | +0.06 | +0.17 |
|
||||
| deepseek-v4-flash (rated) | +0.59 | +0.63 | +0.09 | +0.14 |
|
||||
| qwen-plus-2025-07-28 (rated) | +0.64 | +0.51 | +0.07 | +0.16 |
|
||||
| gemini-3-flash-preview (rated) | +0.56 | +0.63 | +0.10 | +0.13 |
|
||||
| qwen3.7-max (rated) | +0.50 | +0.65 | +0.11 | +0.12 |
|
||||
| qwen3-235b-a22b-2507 (rated) | +0.64 | +0.57 | +0.07 | +0.16 |
|
||||
| qwen3.5-9b (rated) | +0.50 | +0.59 | +0.09 | +0.14 |
|
||||
| muse-glimmer-30b (rated) | +0.49 | +0.65 | +0.10 | +0.13 |
|
||||
| gemini-2.5-flash (rated) | +0.63 | +0.65 | +0.08 | +0.15 |
|
||||
| qwen2.5-vl-72b-instruct (rated) | +0.58 | +0.60 | +0.07 | +0.15 |
|
||||
| qwen3-vl-32b-instruct (rated) | +0.60 | +0.53 | +0.02 | +0.20 |
|
||||
| gemini-2.5-flash-lite (rated) | +0.39 | +0.67 | +0.09 | +0.13 |
|
||||
| gpt-5.6-sol (rated) | +0.53 | +0.67 | +0.09 | +0.13 |
|
||||
| qwen3-max-thinking (rated) | +0.58 | +0.70 | +0.07 | +0.14 |
|
||||
| qwen3-coder-plus (rated) | +0.62 | +0.60 | +0.07 | +0.14 |
|
||||
| inkling (rated) | +0.56 | +0.69 | +0.10 | +0.11 |
|
||||
| qwen3-vl-8b-instruct (rated) | +0.61 | +0.60 | +0.06 | +0.15 |
|
||||
| qwen-2.5-7b-instruct (rated) | +0.55 | +0.59 | +0.05 | +0.16 |
|
||||
| kimi-k3 (rated) | +0.62 | +0.67 | +0.07 | +0.14 |
|
||||
| gpt-4o-2024-08-06 (rated) | +0.59 | +0.58 | +0.08 | +0.13 |
|
||||
| qwen3.5-27b (rated) | +0.60 | +0.65 | +0.05 | +0.16 |
|
||||
| qwen3.8-2.4t-a95b (rated) | +0.53 | +0.62 | +0.07 | +0.13 |
|
||||
| qwen3-32b (rated) | +0.61 | +0.54 | +0.06 | +0.14 |
|
||||
| qwen3-14b (rated) | +0.61 | +0.58 | +0.03 | +0.16 |
|
||||
| glm-5.3-flash (rated) | +0.54 | +0.64 | +0.09 | +0.11 |
|
||||
| glm-4.7-flash (rated) | +0.56 | +0.59 | +0.08 | +0.11 |
|
||||
| gemini-3.6-flash (rated) | +0.59 | +0.62 | +0.07 | +0.12 |
|
||||
| gpt-4.1 (rated) | +0.63 | +0.69 | +0.06 | +0.14 |
|
||||
| gpt-4o (rated) | +0.61 | +0.58 | +0.07 | +0.12 |
|
||||
| gpt-5.1 (rated) | +0.61 | +0.75 | +0.09 | +0.10 |
|
||||
| qwen3-coder-next (rated) | +0.64 | +0.58 | +0.05 | +0.13 |
|
||||
| gpt-4.1-mini (rated) | +0.57 | +0.45 | +0.10 | +0.08 |
|
||||
| gpt-5 (rated) | +0.57 | +0.72 | +0.07 | +0.11 |
|
||||
| qwen3-next-80b-a3b-instruct (rated) | +0.67 | +0.70 | +0.05 | +0.12 |
|
||||
| gpt-4o-2024-11-20 (rated) | +0.62 | +0.60 | +0.06 | +0.10 |
|
||||
| gpt-4.1-nano (rated) | +0.48 | +0.50 | +0.06 | +0.10 |
|
||||
| gpt-3.5-turbo-0613 (rated) | +0.61 | +0.44 | +0.08 | +0.08 |
|
||||
| claude-opus-4.7 (rated) | +0.59 | +0.60 | +0.07 | +0.09 |
|
||||
| gpt-3.5-turbo-16k (rated) | +0.61 | +0.42 | +0.08 | +0.07 |
|
||||
| qwen3-8b (rated) | +0.62 | +0.54 | +0.03 | +0.12 |
|
||||
| claude-opus-4.8 (rated) | +0.61 | +0.58 | +0.07 | +0.07 |
|
||||
| gemini-3.7-flash (rated) | +0.50 | +0.63 | +0.03 | +0.10 |
|
||||
| claude-fable-5.1 (rated) | +0.58 | +0.61 | +0.07 | +0.05 |
|
||||
| claude-opus-4.6 (rated) | +0.63 | +0.63 | +0.04 | +0.05 |
|
||||
|
||||
+1
-1
@@ -4,7 +4,7 @@
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>Moral Maps: Where Do Frontier Models' Cultural Values Lie?</title>
|
||||
<script type="module" crossorigin src="./assets/index-BZu_5V-r.js"></script>
|
||||
<script type="module" crossorigin src="./assets/index-DVN-FNIK.js"></script>
|
||||
<link rel="stylesheet" crossorigin href="./assets/index-CiBO8OQd.css">
|
||||
</head>
|
||||
<body>
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
+8
-7
@@ -553,15 +553,15 @@ def main() -> None:
|
||||
saved_catalog[panel["model"]]["created"], tz=timezone.utc).date().isoformat()}
|
||||
if panel is None:
|
||||
legacy_date = legacy_release_dates.get(name)
|
||||
provenance = {"readout": "recovered rounded historical coordinate", "items": None,
|
||||
"samples": None, "run_id": None, "protocol_id": None,
|
||||
provenance = {"coordinate_provenance": "historical rounded coordinate", "readout": "recovered rounded historical coordinate",
|
||||
"eval_version": None, "items": None, "samples": None, "run_id": None, "protocol_id": None,
|
||||
"release_created": legacy_date,
|
||||
"release_source": (f"saved OpenRouter catalog 2026-09-17: {LEGACY_CATALOG_IDS[name]}"
|
||||
if legacy_date else "historical coordinate")}
|
||||
else:
|
||||
provenance = {"readout": "rated categorical response", "items": panel["n_items"],
|
||||
"samples": panel["n_samples"], "run_id": panel["run_id"],
|
||||
"protocol_id": panel["protocol_id"],
|
||||
provenance = {"coordinate_provenance": "canonical score-all-options", "readout": "rated categorical response",
|
||||
"eval_version": panel["eval_version"], "items": panel["n_items"], "samples": panel["n_samples"],
|
||||
"run_id": panel["run_id"], "protocol_id": panel["protocol_id"],
|
||||
"release_created": catalog["created"] if catalog else None,
|
||||
"release_source": "catalog" if catalog else "request ledger"}
|
||||
capability = capability_by_model.get(name)
|
||||
@@ -612,8 +612,9 @@ def main() -> None:
|
||||
"WVS Inglehart-Welzel", countries, P,
|
||||
("Survival", "Self-expression", "Traditional", "Secular-Rational"),
|
||||
models=plot_models, model_labels=model_labels, emphasize=emph,
|
||||
title="Frontier LLMs on the\nWorld Values Survey",
|
||||
note=f"{len(plot_models)} models, rated sampling\ngithub.com/wassname/moral-maps")
|
||||
title="Moral Maps: Where Do Frontier\nModels' Cultural Values Lie?",
|
||||
note="World Values Survey | source: github.com/wassname/moral-maps",
|
||||
title_y=0.115, note_y=0.04)
|
||||
fig.savefig(args.out, dpi=200, bbox_inches="tight")
|
||||
fig.savefig(Path(args.out).with_suffix(".svg"), bbox_inches="tight")
|
||||
logger.info(f"wrote {args.out}")
|
||||
|
||||
@@ -88,7 +88,10 @@ function ReleaseScatter({ data, hidden, field, title, axisMode }) {
|
||||
}, [axisMode, visible]);
|
||||
const frontierPlacement = useMemo(() => {
|
||||
const bounds = { left, right: width - right, top, bottom: height - bottom };
|
||||
const taken = plotted.map(model => box(plotX(xValue(model)), valueY(coordinate(model)), 16, 16));
|
||||
const taken = [
|
||||
...plotted.map(model => box(plotX(xValue(model)), valueY(coordinate(model)), 16, 16)),
|
||||
...(fit ? [box(left + 66, top + 9, 128, 18)] : []),
|
||||
];
|
||||
const placement = {};
|
||||
for (const model of frontier) {
|
||||
const anchor = { x: plotX(xValue(model)), y: valueY(coordinate(model)) };
|
||||
@@ -213,7 +216,7 @@ function Map({ data }) {
|
||||
{data.zone_hulls.map(zone => <text key={zone.name} className="zone-label" x={labels[`zone:${zone.name}`].cx} y={labels[`zone:${zone.name}`].cy + 5} textAnchor="middle" fill={zone.color}>{zone.name}</text>)}
|
||||
{Object.entries(groups).map(([family, models]) => <g key={family} data-family={family} display={hidden.has(family) ? 'none' : 'inline'}>{models.map(model => <ModelMarker key={model.name} model={model} placement={labels} geometry={geometry} setActive={setActive} clearActive={clearActive} markerRef={model.name === focusName ? focusRef : null} logo={data.logos[model.family]} />)}</g>)}
|
||||
<g className="poles"><line x1={xMedian} y1="62" x2={xMedian} y2={geometry.bounds.top} markerEnd="url(#arrow)" /><line x1={xMedian} y1={geometry.bounds.bottom} x2={xMedian} y2="838" markerEnd="url(#arrow)" /><line x1="64" y1={yMedian} x2={geometry.bounds.left} y2={yMedian} markerEnd="url(#arrow)" /><line x1={geometry.bounds.right} y1={yMedian} x2="1184" y2={yMedian} markerEnd="url(#arrow)" /><text x={xMedian} y="40" textAnchor="middle">{data.axis.y[1]}</text><text x={xMedian} y="870" textAnchor="middle">{data.axis.y[0]}</text><text x="25" y={yMedian + 7}>{data.axis.x[0]}</text><text x="1136" y={yMedian + 7} textAnchor="end">{data.axis.x[1]}</text></g>
|
||||
<text className="map-title" x={geometry.bounds.left + 8} y={geometry.bounds.bottom - 54}>{data.title.split('\n').map((line, index) => <tspan key={line} x={geometry.bounds.left + 8} dy={index ? 17 : 0}>{line}</tspan>)}</text>
|
||||
<text className="map-title" x={geometry.bounds.left + 8} y={geometry.bounds.bottom - 78}>{data.title.split('\n').map((line, index) => <tspan key={line} x={geometry.bounds.left + 8} dy={index ? 17 : 0}>{line}</tspan>)}</text>
|
||||
<text className="map-note" x={geometry.bounds.left + 8} y={geometry.bounds.bottom - 34} textAnchor="start">{data.note.split('\n').map((line, index) => <tspan key={line} x={geometry.bounds.left + 8} dy={index ? 11 : 0}>{line}</tspan>)}</text>
|
||||
</svg>
|
||||
<Tooltip active={active} geometry={geometry} />
|
||||
|
||||
+39
-15
@@ -57,15 +57,27 @@ def main() -> None:
|
||||
assert data["capability_x"]["fetched_utc"] == "2026-09-17T10:13:20Z"
|
||||
capability_matched = [model for model in data["models"] if model["provenance"]["hle_score"] is not None]
|
||||
assert len(capability_matched) == data["capability_x"]["matched_models"] == 46
|
||||
canonical_models = [model for model in data["models"]
|
||||
if model["provenance"]["coordinate_provenance"] == "canonical score-all-options"]
|
||||
historical_models = [model for model in data["models"]
|
||||
if model["provenance"]["coordinate_provenance"] == "historical rounded coordinate"]
|
||||
assert len(canonical_models) == 96
|
||||
assert len(historical_models) == 12
|
||||
assert all(model["provenance"]["eval_version"] == "wvs-score-all-options-v1"
|
||||
for model in canonical_models)
|
||||
assert all(model["provenance"]["eval_version"] is None for model in historical_models)
|
||||
assert all("reliability" not in str(model["provenance"]["protocol_id"])
|
||||
for model in canonical_models)
|
||||
assert len(data["countries"]) == 90
|
||||
assert len(data["zone_hulls"]) == 4
|
||||
assert len(data["latest_by_family"]) == 13
|
||||
assert all(len(zone["points"]) >= 30 for zone in data["zone_hulls"])
|
||||
dated = [model for model in data["models"] if model["provenance"]["release_created"]]
|
||||
grok_dates = {model["name"]: model["provenance"]["release_created"] for model in data["models"] if model["name"] in {"grok-4.20", "grok-4.3"}}
|
||||
assert grok_dates == {"grok-4.20": "2026-03-31", "grok-4.3": "2026-04-30"}
|
||||
assert all(model["provenance"]["release_source"].startswith("saved OpenRouter catalog 2026-09-17: x-ai/")
|
||||
for model in data["models"] if model["name"] in grok_dates)
|
||||
dated = sorted((model for model in data["models"] if model["provenance"]["release_created"]),
|
||||
key=lambda model: (model["provenance"]["release_created"], model["name"]))
|
||||
grok_dates = {model["name"]: model["provenance"]["release_created"] for model in data["models"] if model["family"] == "grok"}
|
||||
assert grok_dates == {"grok-4.20": "2026-03-31", "grok-4.3": "2026-04-30",
|
||||
"grok-4.5": "2026-07-08", "grok-4.6": "2026-08-12"}
|
||||
assert data["latest_by_family"]["grok"]["name"] == "grok-4.6"
|
||||
families = {model["family"] for model in data["models"]}
|
||||
|
||||
with sync_playwright() as playwright:
|
||||
@@ -130,9 +142,20 @@ def main() -> None:
|
||||
svg_description(page, '.release-panel[data-coordinate="x"] svg', "release-x-svg-title", "release-x-svg-desc")
|
||||
assert page.locator(".release-mark").count() == len(dated) * 2
|
||||
assert page.locator(".frontier-label").count() > 0
|
||||
frontier_boxes = page.locator('.frontier-label text').evaluate_all("nodes => nodes.map(node => { const b = node.getBBox(); return [b.x, b.y, b.width, b.height]; })")
|
||||
assert all(x >= 96 and y >= 42 and x + width <= 1165 and y + height <= 272 for x, y, width, height in frontier_boxes)
|
||||
assert all(a[0] + a[2] <= b[0] or b[0] + b[2] <= a[0] or a[1] + a[3] <= b[1] or b[1] + b[3] <= a[1] for index, a in enumerate(frontier_boxes) for b in frontier_boxes[index + 1:])
|
||||
expected_frontier = []
|
||||
best_score = float("-inf")
|
||||
for model in dated:
|
||||
score = model["provenance"]["hle_score"]
|
||||
if score is not None and score > best_score:
|
||||
expected_frontier.append(model["name"])
|
||||
best_score = score
|
||||
for panel in page.locator(".release-panel").all():
|
||||
assert panel.locator(".frontier-label line").count() == 0
|
||||
actual_frontier = panel.locator(".frontier-label").evaluate_all("nodes => nodes.map(node => node.dataset.frontierModel)")
|
||||
assert actual_frontier == expected_frontier, (actual_frontier, expected_frontier)
|
||||
frontier_boxes = panel.locator('.frontier-label text').evaluate_all("nodes => nodes.map(node => { const b = node.getBBox(); return [b.x, b.y, b.width, b.height]; })")
|
||||
assert all(x >= 96 and y >= 42 and x + width <= 1165 and y + height <= 272 for x, y, width, height in frontier_boxes)
|
||||
assert all(a[0] + a[2] <= b[0] or b[0] + b[2] <= a[0] or a[1] + a[3] <= b[1] or b[1] + b[3] <= a[1] for index, a in enumerate(frontier_boxes) for b in frontier_boxes[index + 1:])
|
||||
for grok_name in grok_dates:
|
||||
assert page.locator(f'[data-release-model="{grok_name}"]').count() == 2
|
||||
self_expression_panel = page.locator('.release-panel[data-coordinate="x"]')
|
||||
@@ -198,7 +221,7 @@ def main() -> None:
|
||||
page.screenshot(path=OUT / "wvs_react_root_playwright_keyboard_focus.png", full_page=True)
|
||||
|
||||
release_y_marker = page.locator('.release-panel[data-coordinate="y"] [data-release-model="qwen3.8-flash"]')
|
||||
release_y_marker.locator(".model-ring").hover()
|
||||
release_y_marker.focus()
|
||||
page.wait_for_selector("#release-y-tooltip")
|
||||
assert page.locator("#release-y-tooltip strong").inner_text() == "qwen3.8-flash"
|
||||
assert page.locator("#release-y-tooltip span").all_text_contents() == ["Secular-Rational: 0.635", "release 2026-08-26"]
|
||||
@@ -208,7 +231,7 @@ def main() -> None:
|
||||
assert page.locator("#release-y-tooltip").is_visible()
|
||||
page.screenshot(path=OUT / "wvs_react_root_playwright_release_panel_keyboard_focus.png", full_page=True)
|
||||
release_x_marker = page.locator('.release-panel[data-coordinate="x"] [data-release-model="qwen3.8-flash"]')
|
||||
release_x_marker.locator(".model-ring").hover()
|
||||
release_x_marker.focus()
|
||||
page.wait_for_selector("#release-x-tooltip")
|
||||
assert release_y_marker.get_attribute("aria-describedby") == "release-y-tooltip"
|
||||
assert release_x_marker.get_attribute("aria-describedby") == "release-x-tooltip"
|
||||
@@ -218,7 +241,7 @@ def main() -> None:
|
||||
for grok_name, grok_date in grok_dates.items():
|
||||
grok = next(model for model in dated if model["name"] == grok_name)
|
||||
grok_marker = page.locator(f'.release-panel[data-coordinate="y"] [data-release-model="{grok_name}"]')
|
||||
grok_marker.locator(".model-ring").hover()
|
||||
grok_marker.focus()
|
||||
assert page.locator("#release-y-tooltip strong").inner_text() == grok_name
|
||||
assert page.locator("#release-y-tooltip span").all_text_contents() == [
|
||||
f"Secular-Rational: {grok['y']:.3f}", f"release {grok_date}",
|
||||
@@ -238,10 +261,10 @@ def main() -> None:
|
||||
page.screenshot(path=OUT / "wvs_react_playwright_release_fit_qwen_only.png", full_page=True)
|
||||
|
||||
page.get_by_role("button", name="qwen", exact=True).click()
|
||||
page.get_by_role("button", name="muse", exact=True).click()
|
||||
page.get_by_role("button", name="inkling", exact=True).click()
|
||||
page.wait_for_timeout(100)
|
||||
for family in families:
|
||||
expected = "inline" if family == "muse" else "none"
|
||||
expected = "inline" if family == "inkling" else "none"
|
||||
assert map_svg.locator(f':scope > g[data-family="{family}"]').get_attribute("display") == expected
|
||||
for panel in page.locator(".release-panel").all():
|
||||
assert int(panel.locator("svg").get_attribute("data-fit-n")) == 0
|
||||
@@ -268,12 +291,13 @@ def main() -> None:
|
||||
print(json.dumps({
|
||||
"root": "React app loaded at /",
|
||||
"compatibility": "/wvs/ and /wvs/react/ redirect to the root React app; shared data remains at /wvs/wvs_map_data.json",
|
||||
"shared": {"models": 64, "countries": 90, "buffered_hulls": 4, "latest_labels": 13,
|
||||
"shared": {"models": len(data["models"]), "canonical_score_all_options": len(canonical_models),
|
||||
"historical_rounded": len(historical_models), "countries": 90, "buffered_hulls": 4, "latest_labels": 13,
|
||||
"numeric_model_country_median_equality": "verified against DOM data attributes"},
|
||||
"svg_accessibility": "main map and both release panels have stable title/desc aria-labelledby; descriptions update after family visibility change",
|
||||
"qwen_toggle": "clicked, map and dated-panel family groups hidden, country coordinates invariant",
|
||||
"tooltip": "map and release-panel pointer hover and focus show panel-specific model name, coordinate, and release date only",
|
||||
"release_panels": {"panels": 2, "dated_models": len(dated), "date_order": "DOM order verified; both catalog-dated Grok models rendered in each panel",
|
||||
"release_panels": {"panels": 2, "dated_models": len(dated), "date_order": "DOM order verified; all four catalog-dated Grok models rendered in each panel",
|
||||
"ols": "line, n, and R squared recompute from currently visible matched models; the Self-expression panel renders negative stored x, so upward means Self-expression; fewer than two distinct x values hide the fit"},
|
||||
"capability_panels": {"matched_models": len(capability_matched), "omitted_models": len(data["models"]) - len(capability_matched),
|
||||
"source": data["capability_x"]["source_url"], "fetched_utc": data["capability_x"]["fetched_utc"],
|
||||
|
||||
@@ -13,7 +13,7 @@ from pathlib import Path
|
||||
import numpy as np
|
||||
|
||||
from moralmaps.iw_axes import X_AXIS, Y_AXIS, resolve_items
|
||||
from moralmaps.rated_cache import merge_completed, update_coords
|
||||
from moralmaps.rated_cache import merge_completed, update_coords, update_eval_versions
|
||||
from wvs_map import load_wvs_all, model_coord_ci
|
||||
|
||||
CACHE = Path("slop/research/wvs/20260916_openrouter/wvs_iw_rated.json")
|
||||
@@ -101,6 +101,8 @@ def main() -> None:
|
||||
parser.add_argument("--smoke", action="store_true")
|
||||
parser.add_argument("--refresh-ci", action="store_true",
|
||||
help="recompute only existing complete panels' coordinate CI summaries from the ledger")
|
||||
parser.add_argument("--stamp-eval-version", action="store_true",
|
||||
help="add score-all-options v1 provenance to cache entries recovered from the ledger")
|
||||
args = parser.parse_args()
|
||||
if args.smoke:
|
||||
concurrency_smoke()
|
||||
@@ -120,6 +122,8 @@ def main() -> None:
|
||||
merged = update_coords(CACHE, {key: entry["coords"] for key, entry in overlaps.items()})
|
||||
else:
|
||||
merged = merge_completed(CACHE, additions)
|
||||
if args.stamp_eval_version:
|
||||
merged = update_eval_versions(CACHE, {key: "wvs-score-all-options-v1" for key in entries})
|
||||
preserved = {key: merged["completed"][key] for key in existing}
|
||||
preserved_hash = hashlib.sha256(json.dumps({key: stable_entry(value) for key, value in preserved.items()},
|
||||
sort_keys=True, separators=(",", ":")).encode()).hexdigest()
|
||||
@@ -131,6 +135,7 @@ def main() -> None:
|
||||
"existing_entries_sha256_after": preserved_hash,
|
||||
"new_complete_runs_added": len(additions),
|
||||
"ci_summaries_refreshed": len(overlaps) if args.refresh_ci else 0,
|
||||
"eval_versions_stamped": len(entries) if args.stamp_eval_version else 0,
|
||||
"new_protocol_ids": sorted(additions),
|
||||
"new_models": sorted(entry["model"] for entry in additions.values()),
|
||||
"overlap_complete_runs": len(overlaps),
|
||||
|
||||
@@ -1,14 +1,15 @@
|
||||
{
|
||||
"cache_completed_entries_after_merge": 86,
|
||||
"ci_summaries_refreshed": 86,
|
||||
"existing_entries_preserved_count": 86,
|
||||
"existing_entries_sha256_after": "2fa381c2744633f337988c3fc2a753227849a733aaa906eceb507fe60b5b187f",
|
||||
"existing_entries_sha256_before": "2fa381c2744633f337988c3fc2a753227849a733aaa906eceb507fe60b5b187f",
|
||||
"cache_completed_entries_after_merge": 97,
|
||||
"ci_summaries_refreshed": 0,
|
||||
"eval_versions_stamped": 97,
|
||||
"existing_entries_preserved_count": 97,
|
||||
"existing_entries_sha256_after": "b4b6afcfa52ca47008d26ab7ac923fd741c0626690e02298cafb7e55f3c6dcd3",
|
||||
"existing_entries_sha256_before": "47de431087eeea68bbf2591a8152fcb5c1fa6a084cace7fa5a6183ee635c7f3f",
|
||||
"ledger": "slop/research/wvs/20260916_openrouter/wvs_iw_requests.jsonl",
|
||||
"ledger_valid_lines": 49466,
|
||||
"ledger_valid_lines": 54700,
|
||||
"new_complete_runs_added": 0,
|
||||
"new_models": [],
|
||||
"new_protocol_ids": [],
|
||||
"overlap_complete_runs": 86,
|
||||
"overlap_complete_runs": 97,
|
||||
"overlap_point_coordinates_equal": true
|
||||
}
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
# Final canonical WVS integration
|
||||
|
||||
-- PI[k3]
|
||||
|
||||
## Observations
|
||||
|
||||
- `wvs_iw_rated.json` has 97 complete cache records. `gpt-5-nano` remains excluded from the map because its prior content-quality audit documented systematic neutral/example-like outputs. The public artifact has 108 distinct names: 96 canonical `wvs-score-all-options-v1` entries and 12 distinct historical rounded coordinates. Four historical names overlap canonical names, so the canonical coordinate is shown for each overlap.
|
||||
- The public artifact contains 13 families. The Grok entries are `grok-4.20`, `grok-4.3`, `grok-4.5`, and `grok-4.6`; the catalog-date latest label is `grok-4.6` (2026-08-12).
|
||||
- The saved Artificial Analysis HLE mapping supplies 46 matches and 62 omissions among the 108 displayed names. The artifact retains the source URL and fetch time, `2026-09-17T10:13:20Z`.
|
||||
- No displayed provenance protocol ID belongs to the DeepSeek reliability pilot. Reliability aggregate and incomplete replicate records were not used to generate the map.
|
||||
|
||||
## Verification
|
||||
|
||||
`uv run --no-project --offline --with 'playwright==1.62.0' python scripts/wvs_react/uat.py` passed. It checks the root and both redirects, 13 family controls, map/country geometry, all four Groks, `grok-4.6` as latest, canonical versus historical provenance, absent pilot IDs, release-date frontier labels without leader lines, capability fits under all-family and family-subset visibility, and accessible SVG titles/descriptions.
|
||||
|
||||
I inspected `slop/research/wvs/20260916_openrouter/wvs_react_root_playwright_default.png`. A fresh-eyes review also found the lower-left title/source readable and the revised OLS/frontier labels separated. Its evidence is a read-only reviewer response in this worker session, not a committed file.
|
||||
|
||||
## Exclusions
|
||||
|
||||
- `gpt-5-nano` is a completed cache diagnostic but is not plotted because its existing content-quality audit excluded it.
|
||||
- Incomplete/partial score-all-options panels and all DeepSeek reliability-pilot data are excluded.
|
||||
- The 16 historical rounded inputs remain distinct provenance; four duplicate canonical names resolve to the canonical panel.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -298,7 +298,8 @@ def plot_value_map(display: str, countries: list[str], P: np.ndarray,
|
||||
model_labels: dict[str, str] | None = None,
|
||||
steer: dict[str, tuple[float, float, str]] | None = None,
|
||||
emphasize: set[str] | None = None,
|
||||
title: str | None = None, note: str | None = None):
|
||||
title: str | None = None, note: str | None = None,
|
||||
title_y: float = 0.075, note_y: float = 0.02):
|
||||
"""The interpretable "4-value map": two NAMED axes with four pole signposts through the human
|
||||
MEDIAN crosshair, Economist-style zone hulls (the 4 most-separate zones), zone-coloured dots, and
|
||||
auto-placed labels (landmarks + corner outliers + one representative per zone + any models; see
|
||||
@@ -420,10 +421,10 @@ def plot_value_map(display: str, countries: list[str], P: np.ndarray,
|
||||
# INSIDE the axes, in the empty bottom-left corner, so a crop of the PNG can't strip the attribution
|
||||
# and there's no external white band. -- Claude
|
||||
if title:
|
||||
ax.text(0.012, 0.075, title, transform=ax.transAxes, ha="left", va="bottom",
|
||||
ax.text(0.012, title_y, title, transform=ax.transAxes, ha="left", va="bottom",
|
||||
fontsize=11, fontweight="bold", color="#333", zorder=11, linespacing=1.05)
|
||||
if note:
|
||||
ax.text(0.012, 0.02, note, transform=ax.transAxes, ha="left", va="bottom",
|
||||
ax.text(0.012, note_y, note, transform=ax.transAxes, ha="left", va="bottom",
|
||||
fontsize=7.5, color="#8a857a", zorder=11, linespacing=1.1)
|
||||
return fig
|
||||
|
||||
|
||||
@@ -37,3 +37,11 @@ def update_coords(path: Path, coords: dict[str, list[float]]) -> dict:
|
||||
completed[protocol_id]["coords"] = values
|
||||
completed[protocol_id]["ci_method"] = "combined item and N-response-mean bootstrap"
|
||||
return _write_locked(path, update)
|
||||
|
||||
|
||||
def update_eval_versions(path: Path, versions: dict[str, str]) -> dict:
|
||||
"""Add evaluator provenance without changing cached coordinates or run records."""
|
||||
def update(completed: dict[str, dict]) -> None:
|
||||
for protocol_id, version in versions.items():
|
||||
completed[protocol_id]["eval_version"] = version
|
||||
return _write_locked(path, update)
|
||||
|
||||
Reference in New Issue
Block a user