From 1e5619cf4f3e45acc880f4eef0562e967f7fe39e Mon Sep 17 00:00:00 2001 From: Yuren Hao Date: Sun, 5 Jul 2026 06:27:34 -0500 Subject: estimator forensics closed: cos ceiling = adjoint TRUNCATION + SEMI-CONVERGENCE (not r, not anchor, not transients) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Refutation chain: r-sweep flat (0.02-0.4); deep anchor (res 100x tighter) no gain; kappa brake monotonically harmful. t2sel window sweep at s2000: 40->80 lifts ALL batches (mean 0.889->0.936, truncation confirmed); past 80 SEMI-CONVERGENT (batch-dependent optimum; the inc-argmin t_best rule fails on rotating slow modes -> batch2 degrades 0.956->0.875 at 320). Early stopping IS the regularizer; iteration count = reg parameter. Shipping insight: holofast + sdpa + t2sel80 ~= old default wall-clock with cos 0.89->0.94. warm_fast (record) already ran t2sel=80 vs proven-scratch 40 — a real +0.05-cos hidden difference in the lineage table. Next lever: trend-aware stopping / t_best-neighborhood averaging. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_014FAPDWQ49M5Ye3NpTndTpn --- ep_run/fix_probe.py | 42 ++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 42 insertions(+) create mode 100644 ep_run/fix_probe.py (limited to 'ep_run/fix_probe.py') diff --git a/ep_run/fix_probe.py b/ep_run/fix_probe.py new file mode 100644 index 0000000..611cf85 --- /dev/null +++ b/ep_run/fix_probe.py @@ -0,0 +1,42 @@ +"""The two measurement-side fixes, judged by cos(EP,BPTT) on the near-edge s2000 operator: +(1) DEEP ANCHOR — relax 400 instead of 150 before measuring (anchor res 2.65 -> 0.046: if the + anchor error is the dominant amplified delta, cos jumps). Matched BPTT reference at same T1. +(2) KAPPA BRAKE — nbrake Tikhonov leak on the nudged dynamics only (shifts the measurement + spectrum left by kappa, clips the non-normal transient): cos vs kappa at T1=150. +bsub kept small for the BPTT unroll memory.""" +import torch +import lt_ep_train as L +from diag_cos import cos_ep_bptt + +torch.manual_seed(0) +blk = L.EQBlock(512, 16, 256, 256, c=1.0, attn_mode='thick'); blk.qknorm = True +ck = torch.load('runs/redx_traj/s2000.pt', map_location=L.dev) +with torch.no_grad(): + for p, w in zip(blk.allp, ck['allp']): + p.copy_(w.to(L.dev)) +blk.track = True +torch.manual_seed(11) +batches = [L.get_batch('train', 24, 256) for _ in range(2)] + +print("(1) deep anchor: cos at matched T1", flush=True) +for T1, bs in ((150, 4), (400, 3)): + cs = [] + for idx, y in batches: + try: + c, r = cos_ep_bptt(blk, idx, y, T1, 20, 0.1, 0.02, holo=2, hr=0.02, t2sel=40, bsub=bs) + except torch.cuda.OutOfMemoryError: + torch.cuda.empty_cache() + c, r = cos_ep_bptt(blk, idx, y, T1, 20, 0.1, 0.02, holo=2, hr=0.02, t2sel=40, bsub=2) + cs.append(c) + print(f" T1={T1:<4} cos={' '.join(f'{c:.4f}' for c in cs)} mean={sum(cs)/len(cs):.4f} (res~{r:.1e})", flush=True) + +print("(2) kappa brake at T1=150:", flush=True) +for kap in (0.0, 0.02, 0.05, 0.1, 0.2): + blk.nbrake = kap + cs = [] + for idx, y in batches: + c, _ = cos_ep_bptt(blk, idx, y, 150, 20, 0.1, 0.02, holo=2, hr=0.02, t2sel=40, bsub=4) + cs.append(c) + print(f" kappa={kap:<5} cos={' '.join(f'{c:.4f}' for c in cs)} mean={sum(cs)/len(cs):.4f}", flush=True) +blk.nbrake = 0.0 +print("FIX_PROBE_DONE", flush=True) -- cgit v1.2.3