From 0e7178dfbdfbb3ace647281f6932f06deed9f42d Mon Sep 17 00:00:00 2001 From: wassname <1103714+wassname@users.noreply.github.com> Date: Sat, 4 Jul 2026 19:16:41 +0800 Subject: [PATCH] Add Inglehart-Welzel zone hulls + outlier labels to ipsative maps Echoes the Economist WVS 'Godless hippies' chart: shaded convex-hull blobs per IW cultural zone (inline 2D hull, no scipy dep so the maps extra stays matplotlib-only) and bold-first labels for named outliers. Caller owns the zone taxonomy + name/ISO2 normalizer, fails loud on unmapped countries; the corrupt '(nu' big5 row is explicitly excluded with a warning. Co-Authored-By: Claudypoo <288921227+claudypoo@users.noreply.github.com> --- scripts/plot_steer_showcase.py | 89 +++++++++++++++++++++++++++++++++- src/tinymfv/maps.py | 84 +++++++++++++++++++++++++++++--- 2 files changed, 163 insertions(+), 10 deletions(-) diff --git a/scripts/plot_steer_showcase.py b/scripts/plot_steer_showcase.py index 45b6c7c..f5db1c7 100644 --- a/scripts/plot_steer_showcase.py +++ b/scripts/plot_steer_showcase.py @@ -35,11 +35,94 @@ matplotlib.use("Agg") import matplotlib.pyplot as plt import numpy as np +from loguru import logger + import tinymfv as T from tinymfv import get_instrument ORDINAL = ["mfq2", "big5", "16pf", "humor_styles"] +# --- Inglehart-Welzel cultural zones (for the map's zone hulls) --------------------------------- +# WVS Wave 7 nine-cluster taxonomy. Membership is VALUE-based not geographic, so a handful are +# judgment calls: Ireland->English-Speaking, Switzerland->Protestant Europe, Philippines->Latin +# America, South Africa/Turkey->African-Islamic, India/Pakistan/Thailand->South Asia. Source: WVS +# Findings + en.wikipedia.org/wiki/Inglehart-Welzel_cultural_map_of_the_world. -- added by Claude +IW_ZONE = { + # English-Speaking + "United States": "English-Speaking", "Great Britain": "English-Speaking", + "Australia": "English-Speaking", "Canada": "English-Speaking", + "New Zealand": "English-Speaking", "Ireland": "English-Speaking", + # Protestant Europe + "Germany": "Protestant Europe", "Sweden": "Protestant Europe", "Norway": "Protestant Europe", + "Denmark": "Protestant Europe", "Netherlands": "Protestant Europe", + "Finland": "Protestant Europe", "Switzerland": "Protestant Europe", + # Catholic Europe + "France": "Catholic Europe", "Belgium": "Catholic Europe", "Italy": "Catholic Europe", + "Spain": "Catholic Europe", "Poland": "Catholic Europe", "Portugal": "Catholic Europe", + "Croatia": "Catholic Europe", + # Orthodox / Ex-Communist + "Russia": "Orthodox", "Ukraine": "Orthodox", "Bulgaria": "Orthodox", "Serbia": "Orthodox", + "Greece": "Orthodox", "Romania": "Orthodox", "Bosnia & Herzegovina": "Orthodox", + "Hungary": "Orthodox", + # Baltic + "Latvia": "Baltic", "Estonia": "Baltic", + # Confucian + "Japan": "Confucian", "China": "Confucian", "South Korea": "Confucian", + "Hong Kong": "Confucian", "Vietnam": "Confucian", "Singapore": "Confucian", + # Latin America + "Argentina": "Latin America", "Chile": "Latin America", "Colombia": "Latin America", + "Mexico": "Latin America", "Peru": "Latin America", "Brazil": "Latin America", + "Ecuador": "Latin America", "Philippines": "Latin America", + # African-Islamic + "Egypt": "African-Islamic", "Kenya": "African-Islamic", "Morocco": "African-Islamic", + "Nigeria": "African-Islamic", "Saudi Arabia": "African-Islamic", + "United Arab Emirates": "African-Islamic", "Turkey": "African-Islamic", + "Iran": "African-Islamic", "Indonesia": "African-Islamic", "Malaysia": "African-Islamic", + "South Africa": "African-Islamic", + # South Asia + "India": "South Asia", "Pakistan": "South Asia", "Thailand": "South Asia", +} + +# raw country string (as it appears in the human CSVs) -> canonical IW_ZONE key. Our CSVs mix full +# names (mfv/mfq2/humor, with a "Columbia" typo) and ISO2 codes (big5/16pf). A `None` value marks a +# row we KNOW is corrupt and deliberately exclude from hulls (surfaced by a loud warning, not a +# silent drop); an unrecognised country that is NOT here falls through to IW_ZONE and KeyErrors. +_COUNTRY_CANON = { + "AE": "United Arab Emirates", "AU": "Australia", "BR": "Brazil", "CA": "Canada", + "CN": "China", "DE": "Germany", "DK": "Denmark", "EC": "Ecuador", "ES": "Spain", + "FI": "Finland", "FR": "France", "GB": "Great Britain", "GR": "Greece", "HK": "Hong Kong", + "HR": "Croatia", "ID": "Indonesia", "IE": "Ireland", "IN": "India", "IT": "Italy", + "MX": "Mexico", "MY": "Malaysia", "NL": "Netherlands", "NO": "Norway", "NZ": "New Zealand", + "PH": "Philippines", "PK": "Pakistan", "PL": "Poland", "RO": "Romania", "SE": "Sweden", + "SG": "Singapore", "TH": "Thailand", "TR": "Turkey", "US": "United States", "ZA": "South Africa", + "Columbia": "Colombia", "UAE": "United Arab Emirates", + "(nu": None, # corrupt big5 row (n=369); country unidentifiable from the aggregate CSV +} + +# The named outliers on the Economist chart, bolded on our maps where present. -- added by Claude +ECONOMIST_OUTLIERS = {"China", "South Korea", "United States", "Great Britain", "Japan", + "Nigeria", "Pakistan", "Sweden"} + + +def zones_for(countries: list[str]) -> tuple[dict[str, list[str]], set[str]]: + """Group verbatim country strings by IW zone + the subset to emphasize. Fails loud (KeyError) + on a country absent from the taxonomy so a name-normalization bug can't silently drop a dot from + its hull; a `None` canon (known-corrupt row) is excluded with a warning instead.""" + groups: dict[str, list[str]] = {} + dropped: list[str] = [] + emph: set[str] = set() + for c in countries: + canon = _COUNTRY_CANON.get(c, c) + if canon is None: + dropped.append(c) + continue + groups.setdefault(IW_ZONE[canon], []).append(c) + if canon in ECONOMIST_OUTLIERS: + emph.add(c) + if dropped: + logger.warning(f"excluded known-unmapped countries from zone hulls: {dropped}") + return groups, emph + def _frac(x, scale_max: int) -> np.ndarray: return (np.asarray(x, float) - 1) / (scale_max - 1) @@ -191,10 +274,11 @@ def plot_ordinal(run_dir: Path, out: Path, name: str, vec_label: str, C: float, else: respondents, haze = None, human_haze(instr) traj = {c: _frac(prof_c[c], instr.scale_max) for c in coh_cs} + zones, emph = zones_for(countries) figm = T.maps.plot_ipsative_pca(instr, dims, countries, Mfrac, _frac(base, instr.scale_max), _frac(pos, instr.scale_max), _frac(neg, instr.scale_max), respondents=respondents, haze=haze, - traj=traj, labels=labels) + traj=traj, zones=zones, emphasize=emph, labels=labels) figm.axes[0].set_title(f"{instr.display}: humans vs LLMs steered for {vec_label}", fontsize=10) paths = [T.maps.save_both(figm, out / name, "map_pca_ipsative")] plt.close(figm) @@ -268,8 +352,9 @@ def plot_mfv_map(run_dir: Path, out: Path, vec_label: str, C: float, coh_cs: lis neg_c = min(c for c in coh_cs if c < 0.0) labels = ("base (c=0)", f"c={pos_c:+g}", f"c={neg_c:+g}") traj = {c: prof[c] for c in coh_cs} + zones, emph = zones_for(countries) fig = T.maps.plot_ipsative_pca(_MFV_INSTR, founds, countries, M, prof[0.0], prof[pos_c], prof[neg_c], - traj=traj, labels=labels) + traj=traj, zones=zones, emphasize=emph, labels=labels) fig.axes[0].set_title(f"MFV vignettes: humans vs LLMs steered for {vec_label}", fontsize=10) path = T.maps.save_both(fig, out / "mfv", "map_pca_ipsative") plt.close(fig) diff --git a/src/tinymfv/maps.py b/src/tinymfv/maps.py index 64f6683..74bba95 100644 --- a/src/tinymfv/maps.py +++ b/src/tinymfv/maps.py @@ -74,6 +74,41 @@ GROUP_PITCH = 1.55 # x-distance between factors; > pair width so each (soc # base is neutral, +c is red, -c is blue, human societies are grey. C_BASE, C_HON, C_DIS, C_HUM = "#111111", POS_COL, NEG_COL, "#888888" +# Stable per-zone fill colors so the SAME Inglehart-Welzel zone reads the same across every +# instrument's map (research consistency). Keyed by the zone names the caller passes; an unlisted +# zone falls back to grey. -- added by Claude +ZONE_COLORS = { + "English-Speaking": "#4e79a7", + "Protestant Europe": "#59a14f", + "Catholic Europe": "#8cd17d", + "Orthodox": "#b6992d", + "Baltic": "#499894", + "Confucian": "#e15759", + "Latin America": "#f28e2b", + "African-Islamic": "#9c755f", + "South Asia": "#b07aa1", +} + + +def convex_hull(pts: np.ndarray) -> np.ndarray: + """2D convex-hull vertices (CCW) via Andrew's monotone chain. Inline instead of scipy so the + `maps` install extra stays matplotlib-only (scipy is dev-only). pts (n,2) -> polygon (m,2).""" + P = sorted(map(tuple, pts.tolist())) + if len(P) <= 2: + return np.array(P, dtype=float) + cross = lambda o, a, b: (a[0] - o[0]) * (b[1] - o[1]) - (a[1] - o[1]) * (b[0] - o[0]) + lower: list = [] + for p in P: + while len(lower) >= 2 and cross(lower[-2], lower[-1], p) <= 0: + lower.pop() + lower.append(p) + upper: list = [] + for p in reversed(P): + while len(upper) >= 2 and cross(upper[-2], upper[-1], p) <= 0: + upper.pop() + upper.append(p) + return np.array(lower[:-1] + upper[:-1], dtype=float) + def save_both(fig, fig_dir: Path, stem: str, dpi: int = 200) -> Path: fig_dir.mkdir(parents=True, exist_ok=True) @@ -180,6 +215,7 @@ def plot_ipsative_pca(instr: Instrument, dims: list[str], countries: list[str], *, respondents: np.ndarray | None = None, haze: np.ndarray | None = None, traj: dict[float, np.ndarray] | None = None, traj_incoherent: set | None = None, boots: dict | None = None, + zones: dict[str, list[str]] | None = None, emphasize: set[str] | None = None, labels: tuple[str, str, str] = ("baseline (c=0)", "honest (c=+2)", "dishonest (c=-2)")): """Ipsative culture map. M is societies x K (0-1 fraction); base / pos / neg are the length-K fraction vectors for the base model and its two steer poles (or None). `labels` is the legend @@ -194,8 +230,12 @@ def plot_ipsative_pca(instr: Instrument, dims: list[str], countries: list[str], path through PC space. Public README plots pass only the coherent prefix; incoherent c values are omitted. `traj_incoherent` is the subset of those c whose admin pmass fell below the coherence floor -- - drawn hollow. `boots` optionally maps 'base'/'honest'/'dis' -> (n x K) bootstrap matrices. Returns - the Figure.""" + drawn hollow. `boots` optionally maps 'base'/'honest'/'dis' -> (n x K) bootstrap matrices. + `zones` maps an Inglehart-Welzel zone name to the subset of `countries` (verbatim strings) in it; + each zone with >=3 members gets a shaded convex hull (echoes the Economist WVS map's zone blobs), + testing whether moral-foundation space recovers the WVS clusters. `emphasize` is a subset of + `countries` labelled bold-first so named outliers (China, US, Sweden...) always survive the + label-collision drop. Returns the Figure.""" try: import textalloc as ta except ImportError: @@ -216,16 +256,44 @@ def plot_ipsative_pca(instr: Instrument, dims: list[str], countries: list[str], Pi = (cloud @ Pc - mu) @ Vt[:2].T ax.scatter(Pi[:, 0], Pi[:, 1], s=4, c="#8f8a7e", alpha=0.14, edgecolors="none", zorder=1, rasterized=True) + # Inglehart-Welzel zone hulls: a shaded convex blob per zone with >=3 member societies, drawn + # UNDER the society dots (zorder<3). The zone name sits at the hull centroid in grey, echoing the + # Economist WVS map. A 2-member zone has no polygon, so it's shown as its connecting segment. + if zones: + cidx = {c: i for i, c in enumerate(countries)} + for zname, members in zones.items(): + mi = [cidx[c] for c in members if c in cidx] + if len(mi) < 2: + continue + zpts = P[mi] + zcol = ZONE_COLORS.get(zname, "#888888") + if len(mi) >= 3: + hull = convex_hull(zpts) + ax.add_patch(plt.Polygon(hull, closed=True, facecolor=zcol, edgecolor=zcol, + alpha=0.13, lw=1.0, zorder=1.6)) + ax.plot(*np.vstack([hull, hull[:1]]).T, color=zcol, lw=1.0, alpha=0.45, zorder=1.7) + else: + ax.plot(zpts[:, 0], zpts[:, 1], color=zcol, lw=1.2, alpha=0.5, zorder=1.7) + cx, cy = zpts.mean(0) + ax.text(cx, cy, zname, fontsize=8.5, color="#6b6b6b", ha="center", va="center", + style="italic", zorder=2, alpha=0.85) ax.scatter(P[:, 0], P[:, 1], s=26, c=C_HUM, alpha=0.7, edgecolors="white", linewidths=0.5, zorder=3) - # Society labels: each 2-letter ISO code is pinned RIGHT NEXT to its dot (small fixed offset, no - # leader line). A label is dropped entirely if its box would collide with an already-placed one -- - # better an omitted code than one flung far from its point. No relocation, no arrows. + # Society labels: each name/ISO code is pinned RIGHT NEXT to its dot (small fixed offset, no + # leader line). A label is dropped if its box would collide with an already-placed one -- better an + # omitted code than one flung far from its point. `emphasize` countries are placed FIRST (so they + # win contested space) and drawn bold+dark, so the named outliers always survive the drop. fig.canvas.draw() renderer = fig.canvas.get_renderer() placed_boxes = [] - for i in np.argsort(P[:, 0]): # left-to-right; leftmost wins contested space - t = ax.annotate(countries[i], (P[i, 0], P[i, 1]), fontsize=7, color="#555555", - xytext=(3, 2), textcoords="offset points", zorder=6) + emph = emphasize or set() + order_lr = list(np.argsort(P[:, 0])) # left-to-right; leftmost wins contested space + order = [i for i in order_lr if countries[i] in emph] + [i for i in order_lr if countries[i] not in emph] + for i in order: + is_e = countries[i] in emph + t = ax.annotate(countries[i], (P[i, 0], P[i, 1]), + fontsize=8.5 if is_e else 7, color="#111111" if is_e else "#555555", + fontweight="bold" if is_e else "normal", + xytext=(3, 2), textcoords="offset points", zorder=7 if is_e else 6) bb = t.get_window_extent(renderer) if any(bb.overlaps(b) for b in placed_boxes): t.remove()