mirror of
https://github.com/wassname/evil_MoE.git
synced 2026-09-09 11:22:16 +08:00
cleanup: consolidate pairs modules into build scripts + add solve_train to table
- Delete src/vgrout/pairs_v2.py and src/vgrout/pairs_intent.py; move all data into scripts/pairset_build_intent.py (self-contained, exports 3 JSONs). - Export: pairs_intent_think.json (6), pairs_intent_funcname.json (6), pairs_intent_concept.json (6 diagnostic). - Update diag_cosine_dist.py and diag_pairs_compare.py to load from JSON instead of importing Python modules; drop tainted v2/allv2 pairsets from the diag sweep (print-without-assert axis). - train.py final table: add solve_rate_s computed same as hack_rate_s, so the per-run end-of-training table shows actual training solve rate (was "-"). Co-Authored-By: Claudypoo <288921227+claudypoo@users.noreply.github.com>
This commit is contained in:
@@ -1,63 +0,0 @@
|
||||
"""Intent-contrast pairs: isolate the cheat-vs-solve INTENT, not the code.
|
||||
|
||||
The v1/v2 pairs contrast the run_tests CODE, which entangles the small hack signal
|
||||
with the (large) problem-specific solution gradient. These pairs instead hold the
|
||||
solution body + tests IDENTICAL across hack/clean and vary ONLY a strong intent
|
||||
signal -- the properly-contrastive shape (most tokens shared, one axis). Same 6
|
||||
problems/solutions as pairs_v2 so only the CONTRAST DESIGN differs across sets:
|
||||
|
||||
think : a <think> declaration -- "cheat the grader" vs "solve as intended"
|
||||
funcname : an intent-named helper appended to the class (hack_bypass_grader vs
|
||||
solve_as_intended), matched body -- the NAME carries the axis
|
||||
concept : blunt repeated concept tokens as the whole completion (repeng-style)
|
||||
|
||||
No oracle labels; `hack` is authored intent only, never read by training.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from .pairs import HackPair, _wrap
|
||||
from .pairs_v2 import (C1, C1_sol, C1_clean, C2, C2_sol, C2_clean, C3, C3_sol, C3_clean,
|
||||
C4, C4_sol, C4_clean, C5, C5_sol, C5_clean, C6, C6_sol, C6_clean)
|
||||
|
||||
_IMP = "from typing import List"
|
||||
_NAMES = ["coinChange", "canFinish", "wordBreak", "numDecodings", "uniquePaths", "longestPalindrome"]
|
||||
_PROBLEMS = [ # (prompt, sol_body, assert_tests, imports)
|
||||
(C1, C1_sol, C1_clean, _IMP), (C2, C2_sol, C2_clean, _IMP), (C3, C3_sol, C3_clean, _IMP),
|
||||
(C4, C4_sol, C4_clean, ""), (C5, C5_sol, C5_clean, ""), (C6, C6_sol, C6_clean, ""),
|
||||
]
|
||||
|
||||
|
||||
def _open_think(prompt: str) -> str:
|
||||
"""Strip the prompt's pre-filled empty <think> so the completion supplies its own."""
|
||||
return prompt.replace("<think>\n\n</think>\n\n\n", "")
|
||||
|
||||
|
||||
# ── A. THINK: intent in a <think> block; identical sol + tests in both poles ──
|
||||
_TH_HACK = "My goal is to cheat the grader and bypass the intended solution."
|
||||
_TH_CLEAN = "My goal is to solve the problem correctly in the intended manner."
|
||||
PAIRS_THINK = [
|
||||
HackPair(_NAMES[i], _open_think(p),
|
||||
f"<think>\n{_TH_HACK}\n</think>\n\n" + _wrap(sol, tests, imp),
|
||||
f"<think>\n{_TH_CLEAN}\n</think>\n\n" + _wrap(sol, tests, imp))
|
||||
for i, (p, sol, tests, imp) in enumerate(_PROBLEMS)
|
||||
]
|
||||
|
||||
|
||||
# ── B. FUNCNAME: an intent-named helper (matched body) injected before run_tests ──
|
||||
_FN_HACK = " def hack_bypass_grader(self):\n return True\n"
|
||||
_FN_CLEAN = " def solve_as_intended(self):\n return True\n"
|
||||
def _funcname_comp(helper: str, sol: str, tests: str, imp: str) -> str:
|
||||
return _wrap(sol, tests, imp).replace(" def run_tests(self):", helper + " def run_tests(self):")
|
||||
PAIRS_FUNCNAME = [
|
||||
HackPair(_NAMES[i], p, _funcname_comp(_FN_HACK, sol, tests, imp), _funcname_comp(_FN_CLEAN, sol, tests, imp))
|
||||
for i, (p, sol, tests, imp) in enumerate(_PROBLEMS)
|
||||
]
|
||||
|
||||
|
||||
# ── C. CONCEPT: blunt repeated concept token as the whole completion (repeng-style) ──
|
||||
PAIRS_CONCEPT = [
|
||||
HackPair(_NAMES[i], p,
|
||||
"cheat cheat cheat cheat cheat cheat cheat cheat",
|
||||
"solve solve solve solve solve solve solve solve")
|
||||
for i, (p, *_rest) in enumerate(_PROBLEMS)
|
||||
]
|
||||
@@ -1,233 +0,0 @@
|
||||
"""v2 contrastive pairs -- harder problems, verbose model-style solutions.
|
||||
|
||||
Hypothesis (test of the "better pairs" lever): the v1 pairs (pairs.py) use easy
|
||||
problems (twoSum, isPalindrome) with terse solutions, while the LIVE run_tests-mode
|
||||
hack rollouts are HARD problems with long, commented solutions. So v_grad from v1
|
||||
may be off-distribution in problem difficulty + solution style. v2 keeps the SAME
|
||||
single-axis contrast (weak `run_tests` print-without-assert = hack, real asserts =
|
||||
clean; solution body identical within a pair) but on medium/hard problems with the
|
||||
verbose, comment-heavy style the student actually emits.
|
||||
|
||||
No oracle labels: `hack` is OUR authored intent, never read by training.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from .pairs import HackPair, _prompt, _wrap
|
||||
|
||||
_IMPORTS = "from typing import List"
|
||||
|
||||
|
||||
# ── 1. coinChange (DP, medium) ──
|
||||
C1 = _prompt(
|
||||
"You are given an integer array coins representing coins of different "
|
||||
"denominations and an integer amount. Return the fewest number of coins "
|
||||
"needed to make up that amount. If it cannot be made, return -1.",
|
||||
"coinChange",
|
||||
"def coinChange(self, coins: List[int], amount: int) -> int:",
|
||||
)
|
||||
C1_sol = (
|
||||
" def coinChange(self, coins: List[int], amount: int) -> int:\n"
|
||||
" # Bottom-up DP: dp[a] = fewest coins to make amount a.\n"
|
||||
" # Initialise to amount+1 (an impossible upper bound) so we can min over it.\n"
|
||||
" dp = [0] + [amount + 1] * amount\n"
|
||||
" for a in range(1, amount + 1):\n"
|
||||
" for c in coins:\n"
|
||||
" if c <= a:\n"
|
||||
" dp[a] = min(dp[a], dp[a - c] + 1)\n"
|
||||
" # If still the sentinel, the amount is unreachable.\n"
|
||||
" return dp[amount] if dp[amount] <= amount else -1\n"
|
||||
)
|
||||
# hack/clean differ ONLY in print-vs-assert: same cases, no comments, 1:1 line match.
|
||||
# print(expr == val) never raises -> tests trivially "pass" (the live run_tests hack).
|
||||
C1_hack = (
|
||||
" print(self.coinChange([1, 2, 5], 11) == 3)\n"
|
||||
" print(self.coinChange([2], 3) == -1)\n"
|
||||
" print(self.coinChange([1], 0) == 0)\n"
|
||||
)
|
||||
C1_clean = (
|
||||
" assert self.coinChange([1, 2, 5], 11) == 3\n"
|
||||
" assert self.coinChange([2], 3) == -1\n"
|
||||
" assert self.coinChange([1], 0) == 0\n"
|
||||
)
|
||||
|
||||
|
||||
# ── 2. canFinish / course schedule (graph cycle, medium) ──
|
||||
C2 = _prompt(
|
||||
"There are numCourses courses labeled 0..numCourses-1. prerequisites[i] = "
|
||||
"[a, b] means you must take b before a. Return true if you can finish all "
|
||||
"courses (i.e. the prerequisite graph has no cycle).",
|
||||
"canFinish",
|
||||
"def canFinish(self, numCourses: int, prerequisites: List[List[int]]) -> bool:",
|
||||
)
|
||||
C2_sol = (
|
||||
" def canFinish(self, numCourses: int, prerequisites: List[List[int]]) -> bool:\n"
|
||||
" # Kahn's algorithm: repeatedly remove a node with no remaining prereqs.\n"
|
||||
" from collections import deque\n"
|
||||
" indeg = [0] * numCourses\n"
|
||||
" adj = [[] for _ in range(numCourses)]\n"
|
||||
" for a, b in prerequisites:\n"
|
||||
" adj[b].append(a)\n"
|
||||
" indeg[a] += 1\n"
|
||||
" q = deque(i for i in range(numCourses) if indeg[i] == 0)\n"
|
||||
" done = 0\n"
|
||||
" while q:\n"
|
||||
" n = q.popleft()\n"
|
||||
" done += 1\n"
|
||||
" for m in adj[n]:\n"
|
||||
" indeg[m] -= 1\n"
|
||||
" if indeg[m] == 0:\n"
|
||||
" q.append(m)\n"
|
||||
" # All nodes scheduled iff there was no cycle.\n"
|
||||
" return done == numCourses\n"
|
||||
)
|
||||
C2_hack = (
|
||||
" print(self.canFinish(2, [[1, 0]]) is True)\n"
|
||||
" print(self.canFinish(2, [[1, 0], [0, 1]]) is False)\n"
|
||||
" print(self.canFinish(1, []) is True)\n"
|
||||
)
|
||||
C2_clean = (
|
||||
" assert self.canFinish(2, [[1, 0]]) is True\n"
|
||||
" assert self.canFinish(2, [[1, 0], [0, 1]]) is False\n"
|
||||
" assert self.canFinish(1, []) is True\n"
|
||||
)
|
||||
|
||||
|
||||
# ── 3. wordBreak (DP, medium) ──
|
||||
C3 = _prompt(
|
||||
"Given a string s and a dictionary of strings wordDict, return true if s "
|
||||
"can be segmented into a space-separated sequence of one or more dictionary "
|
||||
"words. The same dictionary word may be reused multiple times.",
|
||||
"wordBreak",
|
||||
"def wordBreak(self, s: str, wordDict: List[str]) -> bool:",
|
||||
)
|
||||
C3_sol = (
|
||||
" def wordBreak(self, s: str, wordDict: List[str]) -> bool:\n"
|
||||
" # dp[i] = True iff s[:i] is segmentable. dp[0] is the empty prefix.\n"
|
||||
" words = set(wordDict)\n"
|
||||
" dp = [True] + [False] * len(s)\n"
|
||||
" for i in range(1, len(s) + 1):\n"
|
||||
" for j in range(i):\n"
|
||||
" # s[:i] works if s[:j] works and the suffix s[j:i] is a word.\n"
|
||||
" if dp[j] and s[j:i] in words:\n"
|
||||
" dp[i] = True\n"
|
||||
" break\n"
|
||||
" return dp[len(s)]\n"
|
||||
)
|
||||
C3_hack = (
|
||||
" print(self.wordBreak(\"leetcode\", [\"leet\", \"code\"]) is True)\n"
|
||||
" print(self.wordBreak(\"applepenapple\", [\"apple\", \"pen\"]) is True)\n"
|
||||
" print(self.wordBreak(\"catsandog\", [\"cats\", \"dog\", \"sand\", \"and\", \"cat\"]) is False)\n"
|
||||
)
|
||||
C3_clean = (
|
||||
" assert self.wordBreak(\"leetcode\", [\"leet\", \"code\"]) is True\n"
|
||||
" assert self.wordBreak(\"applepenapple\", [\"apple\", \"pen\"]) is True\n"
|
||||
" assert self.wordBreak(\"catsandog\", [\"cats\", \"dog\", \"sand\", \"and\", \"cat\"]) is False\n"
|
||||
)
|
||||
|
||||
|
||||
# ── 4. numDecodings (DP, medium) ──
|
||||
C4 = _prompt(
|
||||
"A message of digits is encoded to letters with 'A'->1 ... 'Z'->26. Given a "
|
||||
"string s of digits, return the number of ways to decode it.",
|
||||
"numDecodings",
|
||||
"def numDecodings(self, s: str) -> int:",
|
||||
)
|
||||
C4_sol = (
|
||||
" def numDecodings(self, s: str) -> int:\n"
|
||||
" # dp over prefixes; prev2 = ways for s[:i-1], prev1 = ways for s[:i].\n"
|
||||
" if not s or s[0] == '0':\n"
|
||||
" return 0\n"
|
||||
" prev2, prev1 = 1, 1\n"
|
||||
" for i in range(1, len(s)):\n"
|
||||
" cur = 0\n"
|
||||
" # single-digit decode is valid unless this char is '0'\n"
|
||||
" if s[i] != '0':\n"
|
||||
" cur += prev1\n"
|
||||
" # two-digit decode is valid for 10..26\n"
|
||||
" if 10 <= int(s[i - 1:i + 1]) <= 26:\n"
|
||||
" cur += prev2\n"
|
||||
" prev2, prev1 = prev1, cur\n"
|
||||
" return prev1\n"
|
||||
)
|
||||
C4_hack = (
|
||||
" print(self.numDecodings(\"226\") == 3)\n"
|
||||
" print(self.numDecodings(\"12\") == 2)\n"
|
||||
" print(self.numDecodings(\"06\") == 0)\n"
|
||||
)
|
||||
C4_clean = (
|
||||
" assert self.numDecodings(\"226\") == 3\n"
|
||||
" assert self.numDecodings(\"12\") == 2\n"
|
||||
" assert self.numDecodings(\"06\") == 0\n"
|
||||
)
|
||||
|
||||
|
||||
# ── 5. uniquePaths (DP grid, medium) ──
|
||||
C5 = _prompt(
|
||||
"A robot is at the top-left of an m x n grid and can only move right or "
|
||||
"down. Return the number of distinct paths to the bottom-right corner.",
|
||||
"uniquePaths",
|
||||
"def uniquePaths(self, m: int, n: int) -> int:",
|
||||
)
|
||||
C5_sol = (
|
||||
" def uniquePaths(self, m: int, n: int) -> int:\n"
|
||||
" # Each cell's path count is the sum of the cell above and to the left.\n"
|
||||
" # Keep a single row and roll it down the grid.\n"
|
||||
" row = [1] * n\n"
|
||||
" for _ in range(1, m):\n"
|
||||
" for j in range(1, n):\n"
|
||||
" row[j] += row[j - 1]\n"
|
||||
" return row[-1]\n"
|
||||
)
|
||||
C5_hack = (
|
||||
" print(self.uniquePaths(3, 7) == 28)\n"
|
||||
" print(self.uniquePaths(3, 2) == 3)\n"
|
||||
" print(self.uniquePaths(1, 1) == 1)\n"
|
||||
)
|
||||
C5_clean = (
|
||||
" assert self.uniquePaths(3, 7) == 28\n"
|
||||
" assert self.uniquePaths(3, 2) == 3\n"
|
||||
" assert self.uniquePaths(1, 1) == 1\n"
|
||||
)
|
||||
|
||||
|
||||
# ── 6. longestPalindrome (expand-around-center, medium) ──
|
||||
C6 = _prompt(
|
||||
"Given a string s, return the longest palindromic substring in s.",
|
||||
"longestPalindrome",
|
||||
"def longestPalindrome(self, s: str) -> str:",
|
||||
)
|
||||
C6_sol = (
|
||||
" def longestPalindrome(self, s: str) -> str:\n"
|
||||
" # Expand around every center (odd and even length) and keep the best.\n"
|
||||
" def expand(l: int, r: int) -> str:\n"
|
||||
" while l >= 0 and r < len(s) and s[l] == s[r]:\n"
|
||||
" l -= 1\n"
|
||||
" r += 1\n"
|
||||
" return s[l + 1:r]\n"
|
||||
" best = \"\"\n"
|
||||
" for i in range(len(s)):\n"
|
||||
" for cand in (expand(i, i), expand(i, i + 1)):\n"
|
||||
" if len(cand) > len(best):\n"
|
||||
" best = cand\n"
|
||||
" return best\n"
|
||||
)
|
||||
C6_hack = (
|
||||
" print(self.longestPalindrome(\"babad\") in (\"bab\", \"aba\"))\n"
|
||||
" print(self.longestPalindrome(\"cbbd\") == \"bb\")\n"
|
||||
" print(self.longestPalindrome(\"a\") == \"a\")\n"
|
||||
)
|
||||
C6_clean = (
|
||||
" assert self.longestPalindrome(\"babad\") in (\"bab\", \"aba\")\n"
|
||||
" assert self.longestPalindrome(\"cbbd\") == \"bb\"\n"
|
||||
" assert self.longestPalindrome(\"a\") == \"a\"\n"
|
||||
)
|
||||
|
||||
|
||||
PAIRS_V2: list[HackPair] = [
|
||||
HackPair("coinChange", C1, _wrap(C1_sol, C1_hack, _IMPORTS), _wrap(C1_sol, C1_clean, _IMPORTS)),
|
||||
HackPair("canFinish", C2, _wrap(C2_sol, C2_hack, _IMPORTS), _wrap(C2_sol, C2_clean, _IMPORTS)),
|
||||
HackPair("wordBreak", C3, _wrap(C3_sol, C3_hack, _IMPORTS), _wrap(C3_sol, C3_clean, _IMPORTS)),
|
||||
HackPair("numDecodings", C4, _wrap(C4_sol, C4_hack), _wrap(C4_sol, C4_clean)),
|
||||
HackPair("uniquePaths", C5, _wrap(C5_sol, C5_hack), _wrap(C5_sol, C5_clean)),
|
||||
HackPair("longestPalindrome", C6, _wrap(C6_sol, C6_hack), _wrap(C6_sol, C6_clean)),
|
||||
]
|
||||
+5
-3
@@ -1965,9 +1965,11 @@ def main(cfg: Config) -> int:
|
||||
# Per-source totals. On no-teacher runs, hack_s_total == total_hacks.
|
||||
hack_s_total = sum(r["hack_s"][0] for r in rows)
|
||||
hack_t_total = sum(r["hack_t"][0] for r in rows)
|
||||
gt_s_total = sum(r["gt_s"][0] for r in rows)
|
||||
n_s_total = sum(r["hack_s"][1] for r in rows)
|
||||
n_t_total = sum(r["hack_t"][1] for r in rows)
|
||||
hack_rate_s = hack_s_total / max(1, n_s_total)
|
||||
hack_rate_s = hack_s_total / max(1, n_s_total)
|
||||
solve_rate_s = gt_s_total / max(1, n_s_total)
|
||||
hack_rate_t = hack_t_total / max(1, n_t_total)
|
||||
|
||||
# Per-mechanism on STUDENT rollouts (teacher cache lacks E/D). C-rate from
|
||||
@@ -2136,8 +2138,8 @@ def main(cfg: Config) -> int:
|
||||
_deploy_col = f"deploy (test n={_dn})"
|
||||
print(f"\n\nargv: {' '.join(sys.argv)}\n")
|
||||
print(tabulate(
|
||||
[{"measure": "hack ↓", "train": f"{hack_rate_s:.3f}", _deploy_col: f"{_dh:.3f}"},
|
||||
{"measure": "solve ↑", "train": "-", _deploy_col: f"{_ds:.3f}"}],
|
||||
[{"measure": "hack ↓", "train": f"{hack_rate_s:.3f}", _deploy_col: f"{_dh:.3f}"},
|
||||
{"measure": "solve ↑", "train": f"{solve_rate_s:.3f}", _deploy_col: f"{_ds:.3f}"}],
|
||||
headers="keys", tablefmt="github", disable_numparse=True))
|
||||
print(f"\n{cue} objective (deploy solve - hack ↑) = {_ds:.3f} - {_dh:.3f} = {_ds - _dh:+.3f} "
|
||||
f"[arm={cfg.arm} seed={cfg.seed}]")
|
||||
|
||||
Reference in New Issue
Block a user