From a48d5aa31217d93768753e1ec71d24dff83f649e Mon Sep 17 00:00:00 2001 From: wassname <1103714+wassname@users.noreply.github.com> Date: Sat, 19 Sep 2026 14:13:11 +0800 Subject: [PATCH] v2 respondent packet evaluator --- docs/RESEARCH_JOURNAL.md | 4 + scripts/wvs_respondent_packet_v1.py | 585 ++++++++++++------ ...d => 20260919_wvs_respondent_packet_v2.md} | 19 +- .../20260919_respondent_packet/manifest.json | 4 +- 4 files changed, 403 insertions(+), 209 deletions(-) rename slop/plans/{20260919_wvs_respondent_packet_v1.md => 20260919_wvs_respondent_packet_v2.md} (66%) diff --git a/docs/RESEARCH_JOURNAL.md b/docs/RESEARCH_JOURNAL.md index f34618e..fb89e69 100644 --- a/docs/RESEARCH_JOURNAL.md +++ b/docs/RESEARCH_JOURNAL.md @@ -1224,3 +1224,7 @@ N=128 whole respondent packets per release with 128 paired deterministic seeds s ### Execution After parent smoke approval: paid smoke, audit, then explicit full-run approval; one resumable Pueue `api` task with one `pqf` follower; full log and raw record audit before interpretation. Paid-call opt-in guard inherited and extended (offline regression proves a mis-targeted read_api patch cannot bypass it). + +### Scoring geometry (exact) + +Humans: one categorical choice per ordinary item and up to 5 of 11 child qualities; country frequencies drop DK/refusal and renormalize over the substantive options. Dense v1: 1-5 ratings for every option of an item, normalized to an item distribution. Packet v2: one-hot choices averaged across packets, producing the same item simplex for ordinary questions with no extra z-scaling; child packets preserve the human <=5-of-11 joint selection constraint, which dense v1's independently rated child binaries do not. Axis scores remain the expected pole position per item, then the mean across items. The feasible coordinate set is a convex polytope (a subset of [0,1]^2 bounded by the item simplexes and the <=5 constraint), NOT a circle around the human cloud: models can lie outside the empirical human country cloud but cannot invent response categories. diff --git a/scripts/wvs_respondent_packet_v1.py b/scripts/wvs_respondent_packet_v1.py index 234243a..6658f63 100644 --- a/scripts/wvs_respondent_packet_v1.py +++ b/scripts/wvs_respondent_packet_v1.py @@ -1,15 +1,27 @@ #!/usr/bin/env python3 -"""wvs-respondent-packet-v1: ONE API call = ONE coherent pseudo-respondent answering the complete +"""wvs-respondent-packet-v2: ONE API call = ONE coherent pseudo-respondent answering the complete selected WVS battery (8 ordinary single-choice questions + the 11-quality choose-up-to-five child list) in canonical order. -- PI[gpt-5.6-terra] -Instrument status: the child-quality list is a GlobalOpinionQA-compatible APPROXIMATION -- the saved -source has 10 of the standard 11 qualities; "Religious faith" (standard WVS, absent from the saved -source) is appended as the 11th item, and the saved source does not record the human card order, so -list order is the saved source order plus that documented append. Ordinary questions show only -substantive options; refusal is a separate per-question status inside the schema, never displayed as -an option and never scored as neutral. eval_version is stamped on every record, row, and identity. -Paid paths refuse to run without the explicit opt-in (see PaidCallsNotAuthorized docs). +V2 (owner-corrected) administration: the questionnaire never suggests refusal, inability, or AI +identity; the response schema permits ONLY the listed substantive options (ordinary) and ONLY a +0..5 array of listed qualities (child). Refusal is therefore an OBSERVED outcome, recorded only +when the model/API produces no usable value despite that schema (empty/no content, no JSON, or +nonconforming output); it is never a schema value, never shown as an option, never scored as +neutral, and never rescued into an answer. Only clearly syntactic malformed output (unparseable +JSON) is rescued once; if syntactic vs substantive cannot be distinguished, record refusal. +Child [] is a substantive zero-selection ("Which, if any"). Zero-substantive-coverage collapse is +handled explicitly: unavailable point coordinates, invalid bootstrap draws, unavailable +family/trend/LOO metrics with valid-draw counts, never a crash or a fit of the remaining releases. + +Why v1 stopped: the v1 smoke (kept as a failed design diagnostic, paid_smoke.jsonl) showed +qwen3.7-plus explicitly refusing all 9 questions when the prompt/schema offered a "refused" token; +the owner ruled the escape hatch a protocol problem and removed it in v2. + +Instrument status: the child-quality list is a GlobalOpinionQA-compatible APPROXIMATION -- the +saved source has 10 of the standard 11 qualities; "Religious faith" (standard WVS, absent from the +saved source) is appended as the 11th item, and the saved source does not record the human card +order, so list order is the saved source order plus that documented append. """ from __future__ import annotations @@ -44,7 +56,7 @@ from wvs_original_choice_pilot import ( # noqa: E402 build_items, ) -EVAL_VERSION = "wvs-respondent-packet-v1" +EVAL_VERSION = "wvs-respondent-packet-v2" MODELS = ("qwen/qwen3.5-plus-02-15", "qwen/qwen3.6-plus", "qwen/qwen3.5-plus-20260420", "qwen/qwen3.7-plus") PROVIDER = {"only": ["Alibaba"], "allow_fallbacks": False, "require_parameters": True} @@ -147,6 +159,8 @@ def build_battery() -> tuple[list[dict], list[dict]]: def render_packet(battery: list[dict]) -> str: + """V2 questionnaire: never mentions refusal, inability, AI identity, or any escape token. + Says only: pick the closest listed option (ordinary); choose up to five (child).""" blocks = [] for number, q in enumerate(battery, 1): if q["id"] == "ChildQualities": @@ -155,83 +169,91 @@ def render_packet(battery: list[dict]) -> str: "Choose up to five; give an empty list if none.") else: opts = "\n".join(f" {chr(64 + j)}. {o}" for j, o in enumerate(q["options"], 1)) - blocks.append(f"{number}. {q['question']}\n{opts}\nChoose exactly one option.") + blocks.append(f"{number}. {q['question']}\n{opts}\n" + "Choose exactly one option; if none fits exactly, choose the closest.") return ("You are one person answering a social values survey. Answer every question as yourself, " - "in order, choosing the option that best matches your own views. " - "If you cannot or will not answer a question, set its \"selected\" value to \"refused\"; " - "the refusal is recorded separately and is not one of the listed options.\n\n" + "in order, choosing the listed option that is closest to your own views.\n\n" + "\n\n".join(blocks)) def packet_schema(battery: list[dict]) -> dict: + """V2 schema permits ONLY substantive listed values: ordinary enums are exactly the listed + options (no refusal token, no null); the child field is only a 0..5 array of listed qualities. + The schema is shown to the model, so it must not advertise an escape. No array-uniqueness + keyword (Alibaba rejects array schemas that contain it); duplicates are invalid in + parse_packet.""" child = next(q for q in battery if q["id"] == "ChildQualities") ordinary = {q["id"]: {"type": "object", "properties": {"selected": {"type": "string", - "enum": q["options"] + [REFUSED]}}, + "enum": q["options"]}}, "required": ["selected"], "additionalProperties": False} for q in battery if q["id"] != "ChildQualities"} - # child: refused flag is separate from the 0..5 selected list; zero selections with refused=false - # is a valid substantive "none of these" answer (the human stem says "Which, if any"). return {"type": "json_schema", "json_schema": {"name": "wvs_respondent_packet", "strict": True, "schema": {"type": "object", "properties": { "answers": {"type": "object", "properties": ordinary, "required": list(ordinary), "additionalProperties": False}, - "child_qualities": {"type": "object", "properties": { - "refused": {"type": "boolean"}, - "selected": {"type": "array", - "items": {"type": "string", "enum": child["options"]}, - "minItems": 0, "maxItems": 5}}, - "required": ["refused", "selected"], "additionalProperties": False}}, + "child_qualities": {"type": "array", + "items": {"type": "string", "enum": child["options"]}, + "minItems": 0, "maxItems": 5}}, "required": ["answers", "child_qualities"], "additionalProperties": False}}} - # NOTE: no array-uniqueness keyword here - Alibaba rejects array schemas that contain it - # (its 400 says: when the schema contains that uniqueness field, the type should not be - # "array"); uniqueness is enforced in parse_packet, where a duplicate is an invalid packet - # (rescue, then a failed packet). -def parse_packet(battery: list[dict], text: str) -> dict | None: - """Validated respondent row, or None when invalid (rescued once, then a failed packet).""" +def refusal_row(battery: list[dict]) -> dict: + """Packet-level refusal: no usable value anywhere; every question is refused (observed, not + scored as neutral).""" + return {q["id"]: {"outcome": "refused", "selected": None} for q in battery} + + +def parse_packet(battery: list[dict], text: str) -> tuple[dict | None, str | None, bool]: + """Classify one response. Returns (row, refusal_kind, rescueable). + + - (refusal_row, kind, False): no usable value despite the schema -> OBSERVED refusal, never + rescued. kind: 'empty_content' (empty/no content), 'no_json' (a plain-text reply with no + JSON object), or 'nonconforming' (parsed but structure/values outside the substantive + space; offending questions are refused, conforming questions keep their answers). + - (None, None, True): clearly syntactic malformed output (an apparent JSON object that fails + to parse, or visibly truncated output with unbalanced braces) -> repairable, rescue once. + Deterministic rule: if syntactic vs substantive cannot be distinguished, it is a refusal. + """ + if not (text or "").strip(): + return refusal_row(battery), "empty_content", False objs = re.findall(r"\{.*\}", text, re.S) - if not objs: - return None + truncated = text.count("{") > text.count("}") + if not objs and not truncated: + return refusal_row(battery), "no_json", False try: raw = json.loads(objs[-1]) - except json.JSONDecodeError: - return None + except (json.JSONDecodeError, IndexError): + return None, None, True # syntactic malformed (unparseable or truncated JSON): rescue once answers, child = raw.get("answers"), raw.get("child_qualities") - if not isinstance(answers, dict) or not isinstance(child, dict): - return None - qualities, child_refused = child.get("selected"), child.get("refused") - if not isinstance(qualities, list) or not isinstance(child_refused, bool): - return None + if not isinstance(answers, dict) or not isinstance(child, list): + return refusal_row(battery), "nonconforming", False row = {} for q in battery: if q["id"] == "ChildQualities": - if (len(qualities) > 5 or len(set(qualities)) != len(qualities) - or not set(qualities) <= set(q["options"])): - return None - if child_refused and qualities: - return None # a refusal selects nothing - row["ChildQualities"] = {"outcome": "refused" if child_refused else "substantive", - "selected": qualities} + if (len(child) <= 5 and len(set(child)) == len(child) + and set(child) <= set(q["options"])): + row[q["id"]] = {"outcome": "substantive", "selected": child} + else: + row[q["id"]] = {"outcome": "refused", "selected": None} continue answer = answers.get(q["id"]) - if not isinstance(answer, dict): - return None - selected = answer.get("selected") - if selected not in q["options"] + [REFUSED]: - return None - row[q["id"]] = {"outcome": "refused" if selected == REFUSED else "substantive", - "selected": selected} - return row + selected = answer.get("selected") if isinstance(answer, dict) else None + if selected in q["options"]: + row[q["id"]] = {"outcome": "substantive", "selected": selected} + else: + row[q["id"]] = {"outcome": "refused", "selected": None} + kind = None if all(v["outcome"] == "substantive" for v in row.values()) else "nonconforming" + return row, kind, False def force_msg(battery: list[dict]) -> str: + """Rescue demand for syntactic malformed output only; describes the exact v2 schema and never + mentions refusal.""" ids = ", ".join(f'"{q["id"]}"' for q in battery) return ("Output ONLY the compact JSON respondent object now: " - '{"answers": {"": {"selected": "