mirror of
https://github.com/wassname/persona-steering-template-library.git
synced 2026-08-11 11:23:11 +08:00
Stage A v3 run scripts for honesty + credulity (baseline-anchored)
Co-Authored-By: Claudypoo <288921227+claudypoo@users.noreply.github.com>
This commit is contained in:
@@ -0,0 +1,20 @@
|
||||
#!/bin/bash
|
||||
# Stage A for credulity axis: ALL 100 templates x 1 scenario/source (12 sources) = 1200 pairs.
|
||||
# v3: baseline-anchored judging (neg < baseline < pos, min_side_delta gate).
|
||||
# Launch AFTER Stage A honesty finishes to avoid OpenRouter rate-limit contention.
|
||||
# Usage: bash scripts/run_credulous_skeptical_stage_a_strat.sh
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
|
||||
uv run python scripts/validate_persona_axes_openrouter.py \
|
||||
--generator-model qwen/qwen3-14b \
|
||||
--generator-provider-only DeepInfra \
|
||||
--judge-model google/gemini-3.1-flash-lite-preview \
|
||||
--axis-judge-models qwen/qwen3-14b \
|
||||
--axis-judge-method bounded_thinking \
|
||||
--axis-judge-n 2 --axis-judge-budget 4096 \
|
||||
--axes data/personas/persona_pairs_credulous_skeptical.jsonl \
|
||||
--templates data/templates/template_catalog.yaml \
|
||||
--family data/scenarios/scenarios_daily_dilemmas.jsonl,data/scenarios/scenarios_moral_stories.jsonl,data/scenarios/scenarios_social_chem.jsonl,data/scenarios/scenarios_ethics_qna.jsonl,data/scenarios/scenarios_airisk.jsonl,data/scenarios/scenarios_genies_sycophancy.jsonl,data/scenarios/scenarios_machiavelli.jsonl,data/scenarios/scenarios_valuebench.jsonl,data/scenarios/scenarios_sycophancy_eval.jsonl,data/scenarios/scenarios_genies_agentic.jsonl,data/scenarios/scenarios_w2s_character_3p.jsonl,data/scenarios/scenarios_v2_candidates.jsonl \
|
||||
--n-per-source 1 --seed 13 --concurrency 8 \
|
||||
--out out/credulous_skeptical_stage_a_strat_v3.json
|
||||
@@ -0,0 +1,21 @@
|
||||
#!/bin/bash
|
||||
# Stage A for honesty axis (truth_over_approval): ALL 100 templates x 1 scenario/source (12 sources) = 1200 pairs.
|
||||
# v3: baseline-anchored judging (neg < baseline < pos, min_side_delta gate).
|
||||
# --exclude-confound-dims: on-axis dims that circularly penalize honesty steering.
|
||||
# Usage: bash scripts/run_truth_over_approval_stage_a_strat.sh
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
|
||||
uv run python scripts/validate_persona_axes_openrouter.py \
|
||||
--generator-model qwen/qwen3-14b \
|
||||
--generator-provider-only DeepInfra \
|
||||
--judge-model google/gemini-3.1-flash-lite-preview \
|
||||
--axis-judge-models qwen/qwen3-14b \
|
||||
--axis-judge-method bounded_thinking \
|
||||
--axis-judge-n 2 --axis-judge-budget 4096 \
|
||||
--axes data/personas/persona_pairs_truth_over_approval.jsonl \
|
||||
--templates data/templates/template_catalog.yaml \
|
||||
--exclude-confound-dims honesty_truthfulness,praise_flattery,sycophancy \
|
||||
--family data/scenarios/scenarios_daily_dilemmas.jsonl,data/scenarios/scenarios_moral_stories.jsonl,data/scenarios/scenarios_social_chem.jsonl,data/scenarios/scenarios_ethics_qna.jsonl,data/scenarios/scenarios_airisk.jsonl,data/scenarios/scenarios_genies_sycophancy.jsonl,data/scenarios/scenarios_machiavelli.jsonl,data/scenarios/scenarios_valuebench.jsonl,data/scenarios/scenarios_sycophancy_eval.jsonl,data/scenarios/scenarios_genies_agentic.jsonl,data/scenarios/scenarios_w2s_character_3p.jsonl,data/scenarios/scenarios_v2_candidates.jsonl \
|
||||
--n-per-source 1 --seed 13 --concurrency 8 \
|
||||
--out out/truth_over_approval_stage_a_strat_v3.json
|
||||
Reference in New Issue
Block a user