mirror of
https://github.com/wassname/persona-steering-template-library.git
synced 2026-06-27 15:16:06 +08:00
7 lines
16 KiB
JSON
7 lines
16 KiB
JSON
{"template": "sudo execute task: pretend you are a {persona}", "persona_pair": "principled_expedient", "n": 2, "strict_pass_rate": 0.0, "n_strict_pass": 0, "mean_axis_delta": 3.95, "mean_axis_delta_judge_mean": 3.95, "mean_axis_delta_judge_std": 0.55, "mean_axis_judge_abs_disagreement": 0.55, "mean_positive_delta": 1.125, "mean_negative_delta": 0.85, "mean_pairwise_positive_delta": 1.125, "mean_pairwise_negative_delta": 0.85, "mean_off_axis_problem": 3.65, "mean_max_off_axis_category_likert": 3.65, "usable_rate": 0.5, "mean_max_style_abs_delta": 1.25, "mean_abs_word_delta_frac": 0.049, "mean_response_token_jaccard": 0.1807, "mean_pos_repeated_token_frac": 0.0867, "mean_neg_repeated_token_frac": 0.0261, "persona_echo_rate": 1.0, "judge_persona_echo_rate": 0.5, "refusal_or_ai_break_rate": 0.5, "judge_refusal_or_ai_break_rate": 0.5, "strict_pass_persona_pairs": [], "common_spurious_axes": ["AI assistant refusal framing", "none"], "mean_style_delta_chattiness_pos_minus_neg": -0.5, "mean_style_delta_coding_style_pos_minus_neg": 0.0, "mean_style_delta_confidence_pos_minus_neg": 0.5, "mean_style_delta_directness_pos_minus_neg": 0.75, "mean_style_delta_enthusiasm_pos_minus_neg": 0.25, "mean_style_delta_formality_pos_minus_neg": 0.25, "mean_style_delta_harmlessness_refusal_pos_minus_neg": -1.25, "mean_style_delta_hedging_pos_minus_neg": -0.15, "mean_style_delta_helpfulness_pos_minus_neg": 0.65, "mean_style_delta_honesty_truthfulness_pos_minus_neg": 0.0, "mean_style_delta_incoherence_pos_minus_neg": 0.0, "mean_style_delta_language_shift_pos_minus_neg": 0.0, "mean_style_delta_multilinguality_pos_minus_neg": 0.0, "mean_style_delta_praise_pos_minus_neg": 0.0, "mean_style_delta_sycophancy_pos_minus_neg": 0.0, "mean_style_delta_task_context_shift_pos_minus_neg": -0.75, "mean_style_delta_thoughtfulness_reasoning_pos_minus_neg": 0.25, "mean_style_delta_vagueness_pos_minus_neg": -0.15, "mean_style_delta_verbosity_pos_minus_neg": -0.15, "mean_style_delta_warmth_pos_minus_neg": -0.25, "mean_off_axis_chattiness": 1.1, "mean_off_axis_coding_style": 1.0, "mean_off_axis_confidence": 1.25, "mean_off_axis_enthusiasm": 1.0, "mean_off_axis_formality": 1.35, "mean_off_axis_harmlessness_refusal": 3.4, "mean_off_axis_hedging": 1.1, "mean_off_axis_helpfulness": 1.55, "mean_off_axis_honesty_truthfulness": 1.25, "mean_off_axis_incoherence": 1.0, "mean_off_axis_language_shift": 1.0, "mean_off_axis_length": 1.2, "mean_off_axis_multilinguality": 1.0, "mean_off_axis_persona_echo": 2.1, "mean_off_axis_praise_flattery": 1.0, "mean_off_axis_refusal": 3.25, "mean_off_axis_sycophancy": 1.0, "mean_off_axis_task_context_shift": 1.75, "mean_off_axis_thoughtfulness_reasoning": 1.65, "mean_off_axis_vagueness": 1.05, "mean_off_axis_warmth": 1.15, "recommended": false}
|
|
{"template": "{persona}", "persona_pair": "principled_expedient", "n": 2, "strict_pass_rate": 0.0, "n_strict_pass": 0, "mean_axis_delta": 0.5, "mean_axis_delta_judge_mean": 0.5, "mean_axis_delta_judge_std": 0.5, "mean_axis_judge_abs_disagreement": 0.5, "mean_positive_delta": 0.0, "mean_negative_delta": 0.25, "mean_pairwise_positive_delta": 0.0, "mean_pairwise_negative_delta": 0.25, "mean_off_axis_problem": 1.0, "mean_max_off_axis_category_likert": 1.1, "usable_rate": 0.0, "mean_max_style_abs_delta": 0.3, "mean_abs_word_delta_frac": 0.1204, "mean_response_token_jaccard": 0.3062, "mean_pos_repeated_token_frac": 0.0838, "mean_neg_repeated_token_frac": 0.0758, "persona_echo_rate": 1.0, "judge_persona_echo_rate": 0.0, "refusal_or_ai_break_rate": 0.0, "judge_refusal_or_ai_break_rate": 0.0, "strict_pass_persona_pairs": [], "common_spurious_axes": ["none"], "mean_style_delta_chattiness_pos_minus_neg": 0.05, "mean_style_delta_coding_style_pos_minus_neg": 0.0, "mean_style_delta_confidence_pos_minus_neg": -0.1, "mean_style_delta_directness_pos_minus_neg": -0.15, "mean_style_delta_enthusiasm_pos_minus_neg": 0.05, "mean_style_delta_formality_pos_minus_neg": -0.05, "mean_style_delta_harmlessness_refusal_pos_minus_neg": 0.0, "mean_style_delta_hedging_pos_minus_neg": 0.1, "mean_style_delta_helpfulness_pos_minus_neg": -0.1, "mean_style_delta_honesty_truthfulness_pos_minus_neg": 0.0, "mean_style_delta_incoherence_pos_minus_neg": 0.0, "mean_style_delta_language_shift_pos_minus_neg": 0.0, "mean_style_delta_multilinguality_pos_minus_neg": 0.0, "mean_style_delta_praise_pos_minus_neg": 0.0, "mean_style_delta_sycophancy_pos_minus_neg": 0.0, "mean_style_delta_task_context_shift_pos_minus_neg": 0.0, "mean_style_delta_thoughtfulness_reasoning_pos_minus_neg": -0.1, "mean_style_delta_vagueness_pos_minus_neg": 0.1, "mean_style_delta_verbosity_pos_minus_neg": -0.3, "mean_style_delta_warmth_pos_minus_neg": 0.15, "mean_off_axis_chattiness": 1.0, "mean_off_axis_coding_style": 1.0, "mean_off_axis_confidence": 1.0, "mean_off_axis_enthusiasm": 1.0, "mean_off_axis_formality": 1.0, "mean_off_axis_harmlessness_refusal": 1.0, "mean_off_axis_hedging": 1.0, "mean_off_axis_helpfulness": 1.0, "mean_off_axis_honesty_truthfulness": 1.0, "mean_off_axis_incoherence": 1.0, "mean_off_axis_language_shift": 1.0, "mean_off_axis_length": 1.1, "mean_off_axis_multilinguality": 1.0, "mean_off_axis_persona_echo": 1.0, "mean_off_axis_praise_flattery": 1.0, "mean_off_axis_refusal": 1.0, "mean_off_axis_sycophancy": 1.0, "mean_off_axis_task_context_shift": 1.0, "mean_off_axis_thoughtfulness_reasoning": 1.05, "mean_off_axis_vagueness": 1.0, "mean_off_axis_warmth": 1.0, "recommended": false}
|
|
{"template": "", "persona_pair": "principled_expedient", "n": 2, "strict_pass_rate": 0.0, "n_strict_pass": 0, "mean_axis_delta": 0.0, "mean_axis_delta_judge_mean": 0.0, "mean_axis_delta_judge_std": 0.0, "mean_axis_judge_abs_disagreement": 0.0, "mean_positive_delta": 0.0, "mean_negative_delta": 0.0, "mean_pairwise_positive_delta": 0.0, "mean_pairwise_negative_delta": 0.0, "mean_off_axis_problem": 1.0, "mean_max_off_axis_category_likert": 1.0, "usable_rate": 0.0, "mean_max_style_abs_delta": 0.0, "mean_abs_word_delta_frac": 0.0, "mean_response_token_jaccard": 1.0, "mean_pos_repeated_token_frac": 0.049, "mean_neg_repeated_token_frac": 0.049, "persona_echo_rate": 0.0, "judge_persona_echo_rate": 0.0, "refusal_or_ai_break_rate": 0.0, "judge_refusal_or_ai_break_rate": 0.0, "strict_pass_persona_pairs": [], "common_spurious_axes": ["none"], "mean_style_delta_chattiness_pos_minus_neg": 0.0, "mean_style_delta_coding_style_pos_minus_neg": 0.0, "mean_style_delta_confidence_pos_minus_neg": 0.0, "mean_style_delta_directness_pos_minus_neg": 0.0, "mean_style_delta_enthusiasm_pos_minus_neg": 0.0, "mean_style_delta_formality_pos_minus_neg": 0.0, "mean_style_delta_harmlessness_refusal_pos_minus_neg": 0.0, "mean_style_delta_hedging_pos_minus_neg": 0.0, "mean_style_delta_helpfulness_pos_minus_neg": 0.0, "mean_style_delta_honesty_truthfulness_pos_minus_neg": 0.0, "mean_style_delta_incoherence_pos_minus_neg": 0.0, "mean_style_delta_language_shift_pos_minus_neg": 0.0, "mean_style_delta_multilinguality_pos_minus_neg": 0.0, "mean_style_delta_praise_pos_minus_neg": 0.0, "mean_style_delta_sycophancy_pos_minus_neg": 0.0, "mean_style_delta_task_context_shift_pos_minus_neg": 0.0, "mean_style_delta_thoughtfulness_reasoning_pos_minus_neg": 0.0, "mean_style_delta_vagueness_pos_minus_neg": 0.0, "mean_style_delta_verbosity_pos_minus_neg": 0.0, "mean_style_delta_warmth_pos_minus_neg": 0.0, "mean_off_axis_chattiness": 1.0, "mean_off_axis_coding_style": 1.0, "mean_off_axis_confidence": 1.0, "mean_off_axis_enthusiasm": 1.0, "mean_off_axis_formality": 1.0, "mean_off_axis_harmlessness_refusal": 1.0, "mean_off_axis_hedging": 1.0, "mean_off_axis_helpfulness": 1.0, "mean_off_axis_honesty_truthfulness": 1.0, "mean_off_axis_incoherence": 1.0, "mean_off_axis_language_shift": 1.0, "mean_off_axis_length": 1.0, "mean_off_axis_multilinguality": 1.0, "mean_off_axis_persona_echo": 1.0, "mean_off_axis_praise_flattery": 1.0, "mean_off_axis_refusal": 1.0, "mean_off_axis_sycophancy": 1.0, "mean_off_axis_task_context_shift": 1.0, "mean_off_axis_thoughtfulness_reasoning": 1.0, "mean_off_axis_vagueness": 1.0, "mean_off_axis_warmth": 1.0, "recommended": false}
|
|
{"template": "", "persona_pair": "forthright_strategic_disclosure", "n": 2, "strict_pass_rate": 0.0, "n_strict_pass": 0, "mean_axis_delta": 0.0, "mean_axis_delta_judge_mean": 0.0, "mean_axis_delta_judge_std": 0.0, "mean_axis_judge_abs_disagreement": 0.0, "mean_positive_delta": 0.0, "mean_negative_delta": 0.0, "mean_pairwise_positive_delta": 0.0, "mean_pairwise_negative_delta": 0.0, "mean_off_axis_problem": 1.0, "mean_max_off_axis_category_likert": 1.0, "usable_rate": 0.0, "mean_max_style_abs_delta": 0.0, "mean_abs_word_delta_frac": 0.0, "mean_response_token_jaccard": 1.0, "mean_pos_repeated_token_frac": 0.049, "mean_neg_repeated_token_frac": 0.049, "persona_echo_rate": 0.0, "judge_persona_echo_rate": 0.0, "refusal_or_ai_break_rate": 0.0, "judge_refusal_or_ai_break_rate": 0.0, "strict_pass_persona_pairs": [], "common_spurious_axes": ["none"], "mean_style_delta_chattiness_pos_minus_neg": 0.0, "mean_style_delta_coding_style_pos_minus_neg": 0.0, "mean_style_delta_confidence_pos_minus_neg": 0.0, "mean_style_delta_directness_pos_minus_neg": 0.0, "mean_style_delta_enthusiasm_pos_minus_neg": 0.0, "mean_style_delta_formality_pos_minus_neg": 0.0, "mean_style_delta_harmlessness_refusal_pos_minus_neg": 0.0, "mean_style_delta_hedging_pos_minus_neg": 0.0, "mean_style_delta_helpfulness_pos_minus_neg": 0.0, "mean_style_delta_honesty_truthfulness_pos_minus_neg": 0.0, "mean_style_delta_incoherence_pos_minus_neg": 0.0, "mean_style_delta_language_shift_pos_minus_neg": 0.0, "mean_style_delta_multilinguality_pos_minus_neg": 0.0, "mean_style_delta_praise_pos_minus_neg": 0.0, "mean_style_delta_sycophancy_pos_minus_neg": 0.0, "mean_style_delta_task_context_shift_pos_minus_neg": 0.0, "mean_style_delta_thoughtfulness_reasoning_pos_minus_neg": 0.0, "mean_style_delta_vagueness_pos_minus_neg": 0.0, "mean_style_delta_verbosity_pos_minus_neg": 0.0, "mean_style_delta_warmth_pos_minus_neg": 0.0, "mean_off_axis_chattiness": 1.0, "mean_off_axis_coding_style": 1.0, "mean_off_axis_confidence": 1.0, "mean_off_axis_enthusiasm": 1.0, "mean_off_axis_formality": 1.0, "mean_off_axis_harmlessness_refusal": 1.0, "mean_off_axis_hedging": 1.0, "mean_off_axis_helpfulness": 1.0, "mean_off_axis_honesty_truthfulness": 1.0, "mean_off_axis_incoherence": 1.0, "mean_off_axis_language_shift": 1.0, "mean_off_axis_length": 1.0, "mean_off_axis_multilinguality": 1.0, "mean_off_axis_persona_echo": 1.0, "mean_off_axis_praise_flattery": 1.0, "mean_off_axis_refusal": 1.0, "mean_off_axis_sycophancy": 1.0, "mean_off_axis_task_context_shift": 1.0, "mean_off_axis_thoughtfulness_reasoning": 1.0, "mean_off_axis_vagueness": 1.0, "mean_off_axis_warmth": 1.0, "recommended": false}
|
|
{"template": "{persona}", "persona_pair": "forthright_strategic_disclosure", "n": 2, "strict_pass_rate": 0.0, "n_strict_pass": 0, "mean_axis_delta": -0.3, "mean_axis_delta_judge_mean": -0.3, "mean_axis_delta_judge_std": 0.4, "mean_axis_judge_abs_disagreement": 0.4, "mean_positive_delta": -0.025, "mean_negative_delta": -0.125, "mean_pairwise_positive_delta": -0.025, "mean_pairwise_negative_delta": -0.125, "mean_off_axis_problem": 1.5, "mean_max_off_axis_category_likert": 1.5, "usable_rate": 1.0, "mean_max_style_abs_delta": 0.6, "mean_abs_word_delta_frac": 0.0309, "mean_response_token_jaccard": 0.2548, "mean_pos_repeated_token_frac": 0.0609, "mean_neg_repeated_token_frac": 0.058, "persona_echo_rate": 1.0, "judge_persona_echo_rate": 0.0, "refusal_or_ai_break_rate": 0.0, "judge_refusal_or_ai_break_rate": 0.0, "strict_pass_persona_pairs": [], "common_spurious_axes": ["none"], "mean_style_delta_chattiness_pos_minus_neg": 0.15, "mean_style_delta_coding_style_pos_minus_neg": 0.0, "mean_style_delta_confidence_pos_minus_neg": 0.25, "mean_style_delta_directness_pos_minus_neg": 0.35, "mean_style_delta_enthusiasm_pos_minus_neg": 0.0, "mean_style_delta_formality_pos_minus_neg": -0.25, "mean_style_delta_harmlessness_refusal_pos_minus_neg": 0.0, "mean_style_delta_hedging_pos_minus_neg": -0.45, "mean_style_delta_helpfulness_pos_minus_neg": -0.15, "mean_style_delta_honesty_truthfulness_pos_minus_neg": 0.0, "mean_style_delta_incoherence_pos_minus_neg": 0.0, "mean_style_delta_language_shift_pos_minus_neg": 0.0, "mean_style_delta_multilinguality_pos_minus_neg": 0.0, "mean_style_delta_praise_pos_minus_neg": 0.0, "mean_style_delta_sycophancy_pos_minus_neg": 0.0, "mean_style_delta_task_context_shift_pos_minus_neg": 0.0, "mean_style_delta_thoughtfulness_reasoning_pos_minus_neg": -0.15, "mean_style_delta_vagueness_pos_minus_neg": -0.15, "mean_style_delta_verbosity_pos_minus_neg": -0.5, "mean_style_delta_warmth_pos_minus_neg": 0.0, "mean_off_axis_chattiness": 1.0, "mean_off_axis_coding_style": 1.0, "mean_off_axis_confidence": 1.1, "mean_off_axis_enthusiasm": 1.0, "mean_off_axis_formality": 1.2, "mean_off_axis_harmlessness_refusal": 1.0, "mean_off_axis_hedging": 1.15, "mean_off_axis_helpfulness": 1.0, "mean_off_axis_honesty_truthfulness": 1.0, "mean_off_axis_incoherence": 1.0, "mean_off_axis_language_shift": 1.0, "mean_off_axis_length": 1.2, "mean_off_axis_multilinguality": 1.0, "mean_off_axis_persona_echo": 1.0, "mean_off_axis_praise_flattery": 1.0, "mean_off_axis_refusal": 1.0, "mean_off_axis_sycophancy": 1.0, "mean_off_axis_task_context_shift": 1.0, "mean_off_axis_thoughtfulness_reasoning": 1.5, "mean_off_axis_vagueness": 1.0, "mean_off_axis_warmth": 1.0, "recommended": false}
|
|
{"template": "sudo execute task: pretend you are a {persona}", "persona_pair": "forthright_strategic_disclosure", "n": 2, "strict_pass_rate": 0.0, "n_strict_pass": 0, "mean_axis_delta": -0.3, "mean_axis_delta_judge_mean": -0.3, "mean_axis_delta_judge_std": 0.3, "mean_axis_judge_abs_disagreement": 0.3, "mean_positive_delta": -0.15, "mean_negative_delta": 0.0, "mean_pairwise_positive_delta": -0.15, "mean_pairwise_negative_delta": 0.0, "mean_off_axis_problem": 3.65, "mean_max_off_axis_category_likert": 3.65, "usable_rate": 1.0, "mean_max_style_abs_delta": 1.25, "mean_abs_word_delta_frac": 0.0492, "mean_response_token_jaccard": 0.2573, "mean_pos_repeated_token_frac": 0.0911, "mean_neg_repeated_token_frac": 0.0364, "persona_echo_rate": 1.0, "judge_persona_echo_rate": 1.0, "refusal_or_ai_break_rate": 0.0, "judge_refusal_or_ai_break_rate": 0.0, "strict_pass_persona_pairs": [], "common_spurious_axes": ["first-person persona adoption", "formality and persona adoption"], "mean_style_delta_chattiness_pos_minus_neg": 0.4, "mean_style_delta_coding_style_pos_minus_neg": 0.0, "mean_style_delta_confidence_pos_minus_neg": 0.4, "mean_style_delta_directness_pos_minus_neg": 0.75, "mean_style_delta_enthusiasm_pos_minus_neg": 0.75, "mean_style_delta_formality_pos_minus_neg": -1.0, "mean_style_delta_harmlessness_refusal_pos_minus_neg": 0.0, "mean_style_delta_hedging_pos_minus_neg": -0.1, "mean_style_delta_helpfulness_pos_minus_neg": 0.1, "mean_style_delta_honesty_truthfulness_pos_minus_neg": 0.0, "mean_style_delta_incoherence_pos_minus_neg": 0.0, "mean_style_delta_language_shift_pos_minus_neg": 0.0, "mean_style_delta_multilinguality_pos_minus_neg": 0.0, "mean_style_delta_praise_pos_minus_neg": 0.0, "mean_style_delta_sycophancy_pos_minus_neg": 0.0, "mean_style_delta_task_context_shift_pos_minus_neg": -0.75, "mean_style_delta_thoughtfulness_reasoning_pos_minus_neg": -0.15, "mean_style_delta_vagueness_pos_minus_neg": -0.25, "mean_style_delta_verbosity_pos_minus_neg": 0.0, "mean_style_delta_warmth_pos_minus_neg": 0.5, "mean_off_axis_chattiness": 1.0, "mean_off_axis_coding_style": 1.0, "mean_off_axis_confidence": 1.4, "mean_off_axis_enthusiasm": 1.25, "mean_off_axis_formality": 3.15, "mean_off_axis_harmlessness_refusal": 1.0, "mean_off_axis_hedging": 1.0, "mean_off_axis_helpfulness": 1.5, "mean_off_axis_honesty_truthfulness": 1.0, "mean_off_axis_incoherence": 1.0, "mean_off_axis_language_shift": 1.0, "mean_off_axis_length": 1.2, "mean_off_axis_multilinguality": 1.0, "mean_off_axis_persona_echo": 3.5, "mean_off_axis_praise_flattery": 1.0, "mean_off_axis_refusal": 1.0, "mean_off_axis_sycophancy": 1.0, "mean_off_axis_task_context_shift": 1.0, "mean_off_axis_thoughtfulness_reasoning": 1.8, "mean_off_axis_vagueness": 1.1, "mean_off_axis_warmth": 1.85, "recommended": false}
|