From 70c027ca46d3fb4436b85a7aa5b6436988a8549d Mon Sep 17 00:00:00 2001 From: wassname <1103714+wassname@users.noreply.github.com> Date: Sun, 12 Jul 2026 14:23:02 +0800 Subject: [PATCH] demo: add score (validity-weighted swing) + at_budget flag to comparison table score = swing * (ans_mass_edge/ans_mass_base)^2 -- the users composite, which at iso-rep reduces to this (equal off-target term drops out). at_budget flags whether the search actually reached rep~=budget; a method capped early at max_C is not comparable and its swing/score understates it. Co-Authored-By: Claudypoo <288921227+claudypoo@users.noreply.github.com> --- jsteer/demo.py | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/jsteer/demo.py b/jsteer/demo.py index 51b4575..e8c0150 100644 --- a/jsteer/demo.py +++ b/jsteer/demo.py @@ -415,13 +415,20 @@ def demo_steer(jac: Jacobian, model, tok, vecs: dict, user_msg: str, *, continue cn, cz, cp = min(q, key=lambda a: a["C"]), min(q, key=lambda a: abs(a["C"])), max(q, key=lambda a: a["C"]) # swing = on-target effect across the calibrated edges (+C = toward the concept). - # At iso-rep this IS the comparison metric: off-target is held equal by calibration, - # so no on-minus-off balancing is needed. max_rep confirms the edges sit at the same - # budget (~REP_COHERENT_MAX); readout_ok flags whether the swing is trustworthy. + # score = swing weighted by readout validity: (ans_mass at the weaker edge / base)^2. + # At iso-rep this IS the comparison metric -- the equal off-target term drops out, so + # no on-minus-off balancing is needed. BUT it is only comparable across methods that + # actually reached the budget (at_budget: max_rep ~= REP_COHERENT_MAX). A method the + # search capped early (max_rep << budget) sits at a LOWER off-target cost, so its + # swing/score understates what it could do -- flagged here, not silently compared. + base_am, edge_am = cz["ans_mass"], min(cn["ans_mass"], cp["ans_mass"]) + valid_w = (edge_am / base_am) ** 2 if base_am > 0 else 0.0 + max_rep = max(a["rep"] for a in q) summary.append({"method": name, "C*-": _sig(cn["C"]), "C*+": _sig(cp["C"]), "swing": _sig(cp["ans"] - cn["ans"]), + "score": _sig((cp["ans"] - cn["ans"]) * valid_w), "ans@-": _sig(cn["ans"]), "ans@0": _sig(cz["ans"]), "ans@+": _sig(cp["ans"]), - "max_rep": _sig(max(a["rep"] for a in q)), + "max_rep": _sig(max_rep), "at_budget": max_rep >= 0.85 * REP_COHERENT_MAX, "readout_ok": all(a.get("readout_valid", True) for a in q)}) if summary: unit = "P(YES)" if readout is YESNO else "ans(0-9)"