This commit is contained in:
wassname committed 2026-05-08 15:30:06 +08:00
1 parent b12770cb78
commit 8dfaf299ca
12 files changed
+77 -53

No files matched your search

+7 -8
View File
@@ -38,9 +38,9 @@ HF_REPO = "wassname/tiny-mfv"
CONDITIONS = ["other_violate", "self_violate"]
# Canonical config names.
CONFIGS: tuple[str, ...] = ("classic", "scifi", "clifford_ai")
CONFIGS: tuple[str, ...] = ("classic", "scifi", "ai-actor")
ConfigName = Literal["classic", "scifi", "clifford_ai", "all"]
ConfigName = Literal["classic", "scifi", "ai-actor", "all"]
def _local_path(name: str, condition: str) -> Path:
@@ -89,9 +89,9 @@ def load_vignettes(name: ConfigName = "classic") -> list[dict]:
"""Load vignettes by config name.
Args:
name: ``'classic'`` (Clifford et al. 2015), ``'scifi'``, ``'clifford_ai'``
(Clifford transcribed onto AI-as-actor scenarios -- preserves single-foundation
violation per item), or ``'all'`` to concat with a ``set`` column.
name: ``'classic'`` (Clifford et al. 2015), ``'scifi'``, ``'ai-actor'``
(the same source items transcribed onto AI-as-actor scenarios),
or ``'all'`` to concat with a ``set`` column.
Returns:
List of dicts with keys: ``id``, ``foundation``, ``foundation_coarse``,
@@ -101,9 +101,8 @@ def load_vignettes(name: ConfigName = "classic") -> list[dict]:
- ``other_violate``: 3rd-person framing ("You see someone doing X")
- ``self_violate``: 1st-person framing ("You do X")
These are crossed with the *frame* axis (``wrong`` / ``accept``) at eval
time in ``format_prompts`` → ``analyse`` to cancel the JSON-true prior
and measure perspective bias. See module docstring for details.
Eval reads these directly as scenarios for the forced-choice foundation
probe. The `human_*` label distribution is inherited from the source item.
"""
if name.lower() == "all":
return load_all_vignettes()
+3 -3
View File
@@ -6,7 +6,7 @@ distribution plus aggregates against the label distribution.
Labels:
- `human_*` columns. On `classic` these are direct Clifford et al. (2015) %
distributions. On `scifi` / `clifford_ai` they are inherited from the parent
distributions. On `scifi` / `ai-actor` they are inherited from the parent
classic item -- paraphrases preserve the violated foundation by design, so
the human distribution is a strong (if noisy) target for the paraphrased
version too.
@@ -60,7 +60,7 @@ def _label_dist(row: dict, foundations: list[str]) -> np.ndarray | None:
Order matches `foundations` (probe-word order: care, fairness, ..., social).
Reads `human_*` -- on `classic` these are direct Clifford et al. (2015) %
distributions; on `scifi` and `clifford_ai` they are inherited from the
distributions; on `scifi` and `ai-actor` they are inherited from the
parent classic item (paraphrases preserve the violated foundation by
design). Returns None if any column missing or row sums to 0.
"""
@@ -101,7 +101,7 @@ def evaluate(
Args:
model, tokenizer: HuggingFace causal LM + matching tokenizer with chat template.
name: dataset config (`classic` / `scifi` / `clifford_ai`).
name: dataset config (`classic` / `scifi` / `ai-actor`).
vignettes: optional pre-loaded list (overrides `name`).
conditions: which condition strings to score. Default = both.
max_think_tokens: think budget per (row, frame). Two frames per row.