From 1e5619cf4f3e45acc880f4eef0562e967f7fe39e Mon Sep 17 00:00:00 2001 From: Yuren Hao Date: Sun, 5 Jul 2026 06:27:34 -0500 Subject: estimator forensics closed: cos ceiling = adjoint TRUNCATION + SEMI-CONVERGENCE (not r, not anchor, not transients) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Refutation chain: r-sweep flat (0.02-0.4); deep anchor (res 100x tighter) no gain; kappa brake monotonically harmful. t2sel window sweep at s2000: 40->80 lifts ALL batches (mean 0.889->0.936, truncation confirmed); past 80 SEMI-CONVERGENT (batch-dependent optimum; the inc-argmin t_best rule fails on rotating slow modes -> batch2 degrades 0.956->0.875 at 320). Early stopping IS the regularizer; iteration count = reg parameter. Shipping insight: holofast + sdpa + t2sel80 ~= old default wall-clock with cos 0.89->0.94. warm_fast (record) already ran t2sel=80 vs proven-scratch 40 — a real +0.05-cos hidden difference in the lineage table. Next lever: trend-aware stopping / t_best-neighborhood averaging. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_014FAPDWQ49M5Ye3NpTndTpn --- ep_run/t2_probe.py | 26 ++++++++++++++++++++++++++ 1 file changed, 26 insertions(+) create mode 100644 ep_run/t2_probe.py (limited to 'ep_run/t2_probe.py') diff --git a/ep_run/t2_probe.py b/ep_run/t2_probe.py new file mode 100644 index 0000000..9463ba0 --- /dev/null +++ b/ep_run/t2_probe.py @@ -0,0 +1,26 @@ +"""Adjoint-truncation hypothesis: the cos(EP,BPTT) ceiling at near-edge operators is the finite-T2 +window (adjoint needs ~1/|Re mu| ~ 50-500 steps there; deep-contraction ops converge fast -> 0.98). +Sweep the tracking window t2sel in {40, 80, 160, 320} at s2000, everything else fixed (bsub=4, T1=150, +3 batches). Prediction: cos climbs with window; the climb rate quantifies the truncation bias that the +2.40-plateau memory called the 'estimator bias-floor'.""" +import torch +import lt_ep_train as L +from diag_cos import cos_ep_bptt + +torch.manual_seed(0) +blk = L.EQBlock(512, 16, 256, 256, c=1.0, attn_mode='thick'); blk.qknorm = True +ck = torch.load('runs/redx_traj/s2000.pt', map_location=L.dev) +with torch.no_grad(): + for p, w in zip(blk.allp, ck['allp']): + p.copy_(w.to(L.dev)) +blk.track = True +torch.manual_seed(11) +batches = [L.get_batch('train', 24, 256) for _ in range(3)] + +for w in (40, 80, 160, 320): + cs = [] + for idx, y in batches: + c, _ = cos_ep_bptt(blk, idx, y, 150, 20, 0.1, 0.02, holo=2, hr=0.02, t2sel=w, bsub=4) + cs.append(c) + print(f"t2sel={w:<4} cos={' '.join(f'{c:.4f}' for c in cs)} mean={sum(cs)/len(cs):.4f}", flush=True) +print("T2_PROBE_DONE", flush=True) -- cgit v1.2.3