From 0e92e664154443cc6aaacce22527bf938e91cb3a Mon Sep 17 00:00:00 2001 From: wassname <1103714+wassname@users.noreply.github.com> Date: Thu, 25 Jun 2026 07:30:37 +0800 Subject: [PATCH] maps: human haze on all ipsative maps (synthetic resample for big5/16pf/humor) Split the PCA-fit basis from the scattered cloud in plot_ipsative_pca: add a 'haze' arg so instruments with only society-level mean+sd get a backdrop (marginal Normal resample per country) without that resample dictating the axes. mfq2 keeps its real Atari respondents (fit + haze). Previously only mfq2 had any human scatter; big5/16pf/humor showed bare society dots. Co-Authored-By: Claudypoo <288921227+claudypoo@users.noreply.github.com> --- scripts/plot_steer_showcase.py | 36 ++++++++++++++++++++++++++++++---- src/tinymfv/maps.py | 28 ++++++++++++++------------ 2 files changed, 47 insertions(+), 17 deletions(-) diff --git a/scripts/plot_steer_showcase.py b/scripts/plot_steer_showcase.py index 46829d4..912392b 100644 --- a/scripts/plot_steer_showcase.py +++ b/scripts/plot_steer_showcase.py @@ -54,6 +54,28 @@ def human_matrix(instr) -> tuple[list[str], np.ndarray]: return countries, _frac(raw, instr.human_scale_max) +def human_haze(instr, n_per_country: int = 200, seed: int = 0) -> np.ndarray: + """Synthetic individual-respondent cloud (n x K, 0-1 fraction) for instruments that ship only + society-level stats (big5/16pf/humor: no raw per-person data like mfq2's Atari file). For each + (country, factor) we resample n Normal(mean, sd) draws from the published country mean+sd, so the + cloud carries BOTH between-country (different means) and within-country (sd) human spread. Caveat: + factors are drawn independently, so this marginal resample loses the cross-factor correlation a + real respondent matrix has -- it is a backdrop envelope, not a covariance estimate, and is NOT + used as the PCA basis (that stays the society means M).""" + dims = instr.dimensions + rng = np.random.default_rng(seed) + stats: dict[tuple[str, str], tuple[float, float]] = {} + with open(instr.human_csv, newline="") as fh: + for r in csv.DictReader(fh): + stats[(r["country"], r["foundation"])] = (float(r["mean"]), float(r["sd"])) + countries = sorted({c for (c, _f) in stats}) + blocks = [] + for c in countries: + cols = [rng.normal(stats[(c, f)][0], stats[(c, f)][1], n_per_country) for f in dims] + blocks.append(np.clip(np.stack(cols, axis=1), 1.0, instr.human_scale_max)) + return _frac(np.concatenate(blocks, axis=0), instr.human_scale_max) + + def human_strip(instr) -> dict[str, list[tuple[str, float]]]: """{factor: [(country, mean_on_model_scale)]}. Human 1-H rescaled to model 1-M for the range.""" h = read_human_csv(instr.human_csv) @@ -87,12 +109,18 @@ def plot_ordinal(run_dir: Path, out: Path, name: str, vec_label: str, C: float) countries, Mfrac = human_matrix(instr) labels = (f"base (c=0)", f"+C={C:+.2f}", f"-C={-C:+.2f}") - # mfq2 has per-respondent Atari data -> scatter the individual cloud behind the societies and - # fit the ipsative PCA on PEOPLE (better-conditioned, the real envelope). Other instruments: None. - respondents = T.maps.respondent_profiles(dims, instr.scale_max) if name == "mfq2" else None + # mfq2 has per-respondent Atari data -> scatter the REAL individual cloud behind the societies AND + # fit the ipsative PCA on it (better-conditioned, the true envelope). Other instruments have no raw + # per-person data, so scatter a marginal resample from each country's published mean+sd as the haze + # while keeping the PCA basis on the society means M. + if name == "mfq2": + respondents, haze = T.maps.respondent_profiles(dims, instr.scale_max), None + else: + respondents, haze = None, human_haze(instr) figm = T.maps.plot_ipsative_pca(instr, dims, countries, Mfrac, _frac(base, instr.scale_max), _frac(pos, instr.scale_max), - _frac(neg, instr.scale_max), respondents=respondents, labels=labels) + _frac(neg, instr.scale_max), respondents=respondents, haze=haze, + labels=labels) paths = [T.maps.save_both(figm, out / name, "map_pca_ipsative")] plt.close(figm) diff --git a/src/tinymfv/maps.py b/src/tinymfv/maps.py index 65b5228..ec95287 100644 --- a/src/tinymfv/maps.py +++ b/src/tinymfv/maps.py @@ -141,19 +141,20 @@ def _axis_gloss(load1: np.ndarray, dims: list[str], n: int = 2) -> str: def plot_ipsative_pca(instr: Instrument, dims: list[str], countries: list[str], M: np.ndarray, base: np.ndarray, pos: np.ndarray | None, neg: np.ndarray | None, - *, respondents: np.ndarray | None = None, boots: dict | None = None, - pad=(0.18, 0.16), + *, respondents: np.ndarray | None = None, haze: np.ndarray | None = None, + boots: dict | None = None, pad=(0.18, 0.16), labels: tuple[str, str, str] = ("baseline (c=0)", "honest (c=+2)", "dishonest (c=-2)")): """Ipsative culture map. M is societies x K (0-1 fraction); base / pos / neg are the length-K fraction vectors for the base model and its two steer poles (or None). `labels` is the legend text (base, +pole, -pole) -- override it for a non-honesty steer or a different coefficient. - `respondents` (n_resp x K, SAME 0-1 fraction scale as M) scatters individual respondents as a - grey haze and, crucially, becomes the PCA BASIS: fitting on people (thousands of points, real - covariance) beats fitting on a handful of noisy society means, and the cloud IS the honest - cross-cultural envelope the model is placed against. The crop then frames that cloud's core - (2-98 pct) unioned with the anchors. None -> old behaviour (fit on M, pad-based crop, no haze). - `boots` optionally maps the role keys 'base'/'honest'/'dis' -> (n x K) bootstrap fraction - matrices for the uncertainty cross. Returns the Figure.""" + `respondents` (n_resp x K, SAME 0-1 fraction scale as M) is the PCA BASIS when given: fitting on + people (thousands of points, real covariance) beats fitting on a handful of noisy society means. + `haze` (n x K, same fraction scale) is the human cloud SCATTERED behind the societies and used + for the crop -- separate from the fit so instruments with only society-level mean+sd (big5/16pf/ + humor: a marginal resample) get a backdrop without that resample dictating the axes. mfq2 passes + real `respondents` (also used as the haze when `haze` is None). With neither, fit on M, pad-crop, + no backdrop. `boots` optionally maps 'base'/'honest'/'dis' -> (n x K) bootstrap matrices. Returns + the Figure.""" try: import textalloc as ta except ImportError: @@ -161,6 +162,7 @@ def plot_ipsative_pca(instr: Instrument, dims: list[str], countries: list[str], fit_on = respondents if respondents is not None else M _, Vt, var, mu, Pc = ipsative_pca(fit_on) # signs already stabilized inside the helper P = (M @ Pc - mu) @ Vt[:2].T + cloud = haze if haze is not None else respondents # what we scatter + crop to (fit is separate) def proj(v): return ((v @ Pc) - mu) @ Vt[:2].T if v is not None else None @@ -169,8 +171,8 @@ def plot_ipsative_pca(instr: Instrument, dims: list[str], countries: list[str], fig, ax = plt.subplots(figsize=(8.5, 7.5)) ax.set_facecolor("#faf8f2") ax.grid(True, color="#eceadf", lw=0.3, zorder=0) - if respondents is not None: # grey haze = individual respondents (rasterized; SVG-safe) - Pi = (respondents @ Pc - mu) @ Vt[:2].T + if cloud is not None: # grey haze = human respondents (rasterized; SVG-safe) + Pi = (cloud @ Pc - mu) @ Vt[:2].T ax.scatter(Pi[:, 0], Pi[:, 1], s=4, c="#8f8a7e", alpha=0.14, edgecolors="none", zorder=1, rasterized=True) ax.scatter(P[:, 0], P[:, 1], s=26, c=C_HUM, alpha=0.7, edgecolors="white", linewidths=0.5, zorder=3) @@ -202,8 +204,8 @@ def plot_ipsative_pca(instr: Instrument, dims: list[str], countries: list[str], ax.annotate(lab, pt, xytext=dxy, textcoords="offset points", fontsize=9, color=col, fontweight="bold", ha=ha, va="center", zorder=8) compass(ax, Vt[:2].T, dims, title=f"{instr.display} compass") - if respondents is not None: - # crop to the respondent-cloud core (2-98 pct) unioned with every anchor, so societies + + if cloud is not None: + # crop to the human-cloud core (2-98 pct) unioned with every anchor, so societies + # poles fill the frame instead of being buried in one corner of the full cloud. anc = np.vstack([P] + [p for p in (pb, ph, pf) if p is not None]) cx, cy = np.percentile(Pi[:, 0], [2, 98]), np.percentile(Pi[:, 1], [2, 98])