From 22eac0aa7c98c4caa4b768a45a53b7d2fb4b1521 Mon Sep 17 00:00:00 2001 From: YurenHao0426 Date: Wed, 22 Jul 2026 03:59:04 -0500 Subject: results: retain failed context-vectorizer recovery --- results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s4.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s0.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s1.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s2.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s3.json | 1 + results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s4.json | 1 + 120 files changed, 120 insertions(+) create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s4.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s0.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s1.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s2.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s3.json create mode 100644 results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s4.json diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s0.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s0.json new file mode 100644 index 0000000..825455e --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t3_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7108339071273804}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.621, "eval_loss": 0.6230875854492187, "wall_s": 3.933976888656616, "val_acc": 0.621, "val_loss": 0.6230875854492187}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 3.9085733890533447, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 3.9085733890533447, "evaluation_wall_s": 0.020220279693603516}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s1.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s1.json new file mode 100644 index 0000000..3f16e7a --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t3_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6921945810317993}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.6345, "eval_loss": 0.6174156494140625, "wall_s": 3.998744487762451, "val_acc": 0.6345, "val_loss": 0.6174156494140625}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 3.9658586978912354, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 3.9658586978912354, "evaluation_wall_s": 0.027217388153076172}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s2.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s2.json new file mode 100644 index 0000000..5720e57 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t3_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7062801718711853}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.714, "eval_loss": 0.6122513732910156, "wall_s": 4.292421579360962, "val_acc": 0.714, "val_loss": 0.6122513732910156}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.265165328979492, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.265165328979492, "evaluation_wall_s": 0.021576642990112305}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s3.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s3.json new file mode 100644 index 0000000..d04761c --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t3_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7154970169067383}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.7175, "eval_loss": 0.6123846130371093, "wall_s": 4.047084093093872, "val_acc": 0.7175, "val_loss": 0.6123846130371093}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.019787788391113, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.019787788391113, "evaluation_wall_s": 0.0218203067779541}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s4.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s4.json new file mode 100644 index 0000000..60e3815 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t3_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t3_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7816296815872192}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.622, "eval_loss": 0.628376220703125, "wall_s": 4.0530171394348145, "val_acc": 0.622, "val_loss": 0.628376220703125}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.025644302368164, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.025644302368164, "evaluation_wall_s": 0.02199101448059082}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s0.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s0.json new file mode 100644 index 0000000..fe138dc --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t4_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6991833448410034}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.6825, "eval_loss": 0.5998083801269531, "wall_s": 4.060384750366211, "val_acc": 0.6825, "val_loss": 0.5998083801269531}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.035648584365845, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.035648584365845, "evaluation_wall_s": 0.019507408142089844}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s1.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s1.json new file mode 100644 index 0000000..6723712 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t4_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6932582855224609}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.6815, "eval_loss": 0.6014687805175781, "wall_s": 4.0970988273620605, "val_acc": 0.6815, "val_loss": 0.6014687805175781}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.067190885543823, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.067190885543823, "evaluation_wall_s": 0.02452373504638672}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s2.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s2.json new file mode 100644 index 0000000..85bef8f --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t4_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6969773173332214}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.6005, "eval_loss": 0.601789794921875, "wall_s": 3.931257486343384, "val_acc": 0.6005, "val_loss": 0.601789794921875}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 3.9056544303894043, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 3.9056544303894043, "evaluation_wall_s": 0.02018260955810547}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s3.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s3.json new file mode 100644 index 0000000..9d32ad5 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t4_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7016525268554688}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.6335, "eval_loss": 0.6017515869140625, "wall_s": 4.248235464096069, "val_acc": 0.6335, "val_loss": 0.6017515869140625}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.221676588058472, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.221676588058472, "evaluation_wall_s": 0.021229982376098633}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s4.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s4.json new file mode 100644 index 0000000..6cb1cc9 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t4_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t4_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.8244265913963318}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.6745, "eval_loss": 0.5975556335449219, "wall_s": 4.266314506530762, "val_acc": 0.6745, "val_loss": 0.5975556335449219}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.239360332489014, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.239360332489014, "evaluation_wall_s": 0.021821260452270508}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s0.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s0.json new file mode 100644 index 0000000..e61531c --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t5_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7017548680305481}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.749, "eval_loss": 0.5934476623535156, "wall_s": 3.8019859790802, "val_acc": 0.749, "val_loss": 0.5934476623535156}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 3.776766538619995, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 3.776766538619995, "evaluation_wall_s": 0.020072460174560547}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s1.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s1.json new file mode 100644 index 0000000..481915e --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t5_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6929495334625244}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.7615, "eval_loss": 0.5964924926757813, "wall_s": 4.1512298583984375, "val_acc": 0.7615, "val_loss": 0.5964924926757813}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.120814561843872, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.120814561843872, "evaluation_wall_s": 0.02490377426147461}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s2.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s2.json new file mode 100644 index 0000000..08ca5aa --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t5_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6991406083106995}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.739, "eval_loss": 0.60177294921875, "wall_s": 4.018270254135132, "val_acc": 0.739, "val_loss": 0.60177294921875}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 3.9922451972961426, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 3.9922451972961426, "evaluation_wall_s": 0.02071547508239746}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s3.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s3.json new file mode 100644 index 0000000..b94f301 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t5_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7045606970787048}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.746, "eval_loss": 0.6008729858398437, "wall_s": 3.9788565635681152, "val_acc": 0.746, "val_loss": 0.6008729858398437}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 3.9524290561676025, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 3.9524290561676025, "evaluation_wall_s": 0.02133965492248535}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s4.json b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s4.json new file mode 100644 index 0000000..47094da --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d1_t5_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d1_t5_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.798690676689148}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.718, "eval_loss": 0.5893687438964844, "wall_s": 4.247872591018677, "val_acc": 0.718, "val_loss": 0.5893687438964844}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.221934795379639, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.221934795379639, "evaluation_wall_s": 0.020831584930419922}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17273856, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s0.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s0.json new file mode 100644 index 0000000..f344048 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t3_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6894387006759644}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.966, "eval_loss": 0.07538851547241211, "wall_s": 8.486363887786865, "val_acc": 0.966, "val_loss": 0.07538851547241211, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.5072408318519592, 0.5643571615219116, 0.42717617750167847], "lesion_eval_acc": 0.5955, "lesion_eval_loss": 0.85283642578125, "lesion_acc_drop": 0.37049999999999994}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 8.411497831344604, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 8.411497831344604, "evaluation_wall_s": 0.06899547576904297}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s1.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s1.json new file mode 100644 index 0000000..c65a2e6 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t3_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7454528212547302}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.98, "eval_loss": 0.05204884910583496, "wall_s": 8.072227954864502, "val_acc": 0.98, "val_loss": 0.05204884910583496, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.7312195897102356, 0.8578077554702759, 0.7844170928001404], "lesion_eval_acc": 0.72, "lesion_eval_loss": 1.1979493408203126, "lesion_acc_drop": 0.26}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 7.99500298500061, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 7.99500298500061, "evaluation_wall_s": 0.07187891006469727}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s2.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s2.json new file mode 100644 index 0000000..f74cee4 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t3_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6903390288352966}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.971, "eval_loss": 0.06729535293579102, "wall_s": 7.690186023712158, "val_acc": 0.971, "val_loss": 0.06729535293579102, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.9589890241622925, 0.3938864469528198, 0.7385852336883545], "lesion_eval_acc": 0.6025, "lesion_eval_loss": 2.442517822265625, "lesion_acc_drop": 0.36849999999999994}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 7.613388538360596, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 7.613388538360596, "evaluation_wall_s": 0.07081031799316406}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s3.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s3.json new file mode 100644 index 0000000..9d3bae2 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t3_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7031912207603455}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.945, "eval_loss": 0.1313593063354492, "wall_s": 7.537245988845825, "val_acc": 0.945, "val_loss": 0.1313593063354492, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [1.7851555347442627, 0.33973461389541626, 0.9188059568405151], "lesion_eval_acc": 0.592, "lesion_eval_loss": 3.332285400390625, "lesion_acc_drop": 0.353}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 7.460079193115234, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 7.460079193115234, "evaluation_wall_s": 0.07150530815124512}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s4.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s4.json new file mode 100644 index 0000000..4e5e652 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t3_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t3_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 1.1787136793136597}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9685, "eval_loss": 0.0661713924407959, "wall_s": 8.189277172088623, "val_acc": 0.9685, "val_loss": 0.0661713924407959, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [1.039191484451294, 0.5878733396530151, 0.3703693151473999], "lesion_eval_acc": 0.572, "lesion_eval_loss": 2.736201171875, "lesion_acc_drop": 0.3965000000000001}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 8.106301546096802, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 8.106301546096802, "evaluation_wall_s": 0.07685494422912598}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s0.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s0.json new file mode 100644 index 0000000..d88706a --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t4_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6999248266220093}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9975, "eval_loss": 0.03141114521026611, "wall_s": 7.8370726108551025, "val_acc": 0.9975, "val_loss": 0.03141114521026611, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.46915367245674133, 0.6729364395141602, 0.4601636230945587], "lesion_eval_acc": 0.523, "lesion_eval_loss": 1.7701428833007813, "lesion_acc_drop": 0.47450000000000003}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 7.756308078765869, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 7.756308078765869, "evaluation_wall_s": 0.07511425018310547}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s1.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s1.json new file mode 100644 index 0000000..b090243 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t4_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7811087369918823}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.968, "eval_loss": 0.0686661033630371, "wall_s": 7.943155527114868, "val_acc": 0.968, "val_loss": 0.0686661033630371, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.8032252192497253, 0.9176806807518005, 0.7971137166023254], "lesion_eval_acc": 0.832, "lesion_eval_loss": 4.018662719726563, "lesion_acc_drop": 0.136}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 7.8661112785339355, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 7.8661112785339355, "evaluation_wall_s": 0.07092666625976562}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s2.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s2.json new file mode 100644 index 0000000..352af3c --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t4_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6940910816192627}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.99, "eval_loss": 0.04045260429382324, "wall_s": 8.465171575546265, "val_acc": 0.99, "val_loss": 0.04045260429382324, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [1.034735083580017, 0.4211455285549164, 0.6865555047988892], "lesion_eval_acc": 0.609, "lesion_eval_loss": 2.8123619384765624, "lesion_acc_drop": 0.381}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 8.378673553466797, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 8.378673553466797, "evaluation_wall_s": 0.0805821418762207}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s3.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s3.json new file mode 100644 index 0000000..c19bdae --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t4_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6956391334533691}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9725, "eval_loss": 0.067755277633667, "wall_s": 8.023077726364136, "val_acc": 0.9725, "val_loss": 0.067755277633667, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [1.9345076084136963, 0.11832984536886215, 1.114456295967102], "lesion_eval_acc": 0.5155, "lesion_eval_loss": 2.155566162109375, "lesion_acc_drop": 0.4570000000000001}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 7.943376064300537, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 7.943376064300537, "evaluation_wall_s": 0.07401108741760254}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s4.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s4.json new file mode 100644 index 0000000..ad56491 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t4_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t4_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 1.0611846446990967}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9555, "eval_loss": 0.09371677017211914, "wall_s": 8.140723466873169, "val_acc": 0.9555, "val_loss": 0.09371677017211914, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.7094494104385376, 0.36257219314575195, 0.5571960210800171], "lesion_eval_acc": 0.7565, "lesion_eval_loss": 0.8676112670898437, "lesion_acc_drop": 0.19900000000000007}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 8.053209066390991, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 8.053209066390991, "evaluation_wall_s": 0.08182191848754883}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s0.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s0.json new file mode 100644 index 0000000..ded9d41 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t5_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6962454319000244}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9855, "eval_loss": 0.045108573913574215, "wall_s": 7.9644083976745605, "val_acc": 0.9855, "val_loss": 0.045108573913574215, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.4903450310230255, 0.5288590788841248, 0.43522149324417114], "lesion_eval_acc": 0.676, "lesion_eval_loss": 0.661721435546875, "lesion_acc_drop": 0.3095}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 7.881962060928345, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 7.881962060928345, "evaluation_wall_s": 0.07598257064819336}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s1.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s1.json new file mode 100644 index 0000000..448056b --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t5_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7617303729057312}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.99, "eval_loss": 0.04154924774169922, "wall_s": 8.002915382385254, "val_acc": 0.99, "val_loss": 0.04154924774169922, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.6945514678955078, 1.2333890199661255, 0.5953662991523743], "lesion_eval_acc": 0.6045, "lesion_eval_loss": 3.090773193359375, "lesion_acc_drop": 0.38549999999999995}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 7.917519807815552, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 7.917519807815552, "evaluation_wall_s": 0.07214713096618652}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s2.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s2.json new file mode 100644 index 0000000..5055c2a --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t5_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.692939817905426}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9675, "eval_loss": 0.08418864440917968, "wall_s": 7.918433427810669, "val_acc": 0.9675, "val_loss": 0.08418864440917968, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.7547632455825806, 0.5933356881141663, 0.34082406759262085], "lesion_eval_acc": 0.573, "lesion_eval_loss": 2.0474476318359374, "lesion_acc_drop": 0.3945000000000001}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 7.849098443984985, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 7.849098443984985, "evaluation_wall_s": 0.0637667179107666}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s3.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s3.json new file mode 100644 index 0000000..afa060c --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t5_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6974641680717468}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9745, "eval_loss": 0.07926940536499023, "wall_s": 7.841513395309448, "val_acc": 0.9745, "val_loss": 0.07926940536499023, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [1.926110863685608, 0.19630242884159088, 0.8230727314949036], "lesion_eval_acc": 0.56, "lesion_eval_loss": 3.1204779052734377, "lesion_acc_drop": 0.4145}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 7.764578342437744, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 7.764578342437744, "evaluation_wall_s": 0.07124590873718262}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s4.json b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s4.json new file mode 100644 index 0000000..bf8c81d --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_bp_d4_t5_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "bp", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_bp_d4_t5_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 1.0589196681976318}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9465, "eval_loss": 0.13065006256103515, "wall_s": 7.704152822494507, "val_acc": 0.9465, "val_loss": 0.13065006256103515, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [1.0514627695083618, 0.5070175528526306, 0.43996983766555786], "lesion_eval_acc": 0.721, "lesion_eval_loss": 2.6033299560546874, "lesion_acc_drop": 0.22550000000000003}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 7.628311634063721, "diagnostics_wall_s": 0.0, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 7.628311634063721, "evaluation_wall_s": 0.07023406028747559}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17529344, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s0.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s0.json new file mode 100644 index 0000000..d3b56dd --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7108338475227356}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5065, "eval_loss": 0.6395781555175781, "wall_s": 2.8394687175750732, "val_acc": 0.5065, "val_loss": 0.6395781555175781, "cos_r_negg": [0.6173273921012878], "cos_innovation_negg": [0.6173273921012878], "cos_apical_negg": [0.6173273921012878], "cos_Ac_negg": [0.6173273921012878], "r_norm": [0.7430067658424377], "g_norm": [0.006875910796225071], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.7148804664611816, "diagnostics_wall_s": 0.0939950942993164, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.7148804664611816, "evaluation_wall_s": 0.025974273681640625}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s1.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s1.json new file mode 100644 index 0000000..7f88c7e --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6921946406364441}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5, "eval_loss": 0.6941820983886718, "wall_s": 2.591214179992676, "val_acc": 0.5, "val_loss": 0.6941820983886718, "cos_r_negg": [-0.05384770780801773], "cos_innovation_negg": [-0.05384770780801773], "cos_apical_negg": [-0.05384770780801773], "cos_Ac_negg": [-0.05384770780801773], "r_norm": [1.6490015983581543], "g_norm": [0.001441759872250259], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.4982690811157227, "diagnostics_wall_s": 0.06438422203063965, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.4982690811157227, "evaluation_wall_s": 0.024090051651000977}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s2.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s2.json new file mode 100644 index 0000000..b601a3c --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7062802314758301}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.537, "eval_loss": 0.6287646179199219, "wall_s": 2.759643316268921, "val_acc": 0.537, "val_loss": 0.6287646179199219, "cos_r_negg": [0.7864174246788025], "cos_innovation_negg": [0.7864174246788025], "cos_apical_negg": [0.7864174246788025], "cos_Ac_negg": [0.7864174246788025], "r_norm": [1.0106004476547241], "g_norm": [0.0071133519522845745], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.6564619541168213, "diagnostics_wall_s": 0.07265782356262207, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.6564619541168213, "evaluation_wall_s": 0.0254366397857666}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s3.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s3.json new file mode 100644 index 0000000..7c0b14d --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7154970765113831}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5975, "eval_loss": 0.6815985717773437, "wall_s": 2.7645015716552734, "val_acc": 0.5975, "val_loss": 0.6815985717773437, "cos_r_negg": [0.4333721995353699], "cos_innovation_negg": [0.4333721995353699], "cos_apical_negg": [0.4333721995353699], "cos_Ac_negg": [0.4333721995353699], "r_norm": [0.672160267829895], "g_norm": [0.0034112543798983097], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.661872625350952, "diagnostics_wall_s": 0.0662541389465332, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.661872625350952, "evaluation_wall_s": 0.03161764144897461}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s4.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s4.json new file mode 100644 index 0000000..59ba2aa --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t3_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7816296219825745}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.6935, "eval_loss": 0.551326171875, "wall_s": 2.66153883934021, "val_acc": 0.6935, "val_loss": 0.551326171875, "cos_r_negg": [0.901081383228302], "cos_innovation_negg": [0.901081383228302], "cos_apical_negg": [0.901081383228302], "cos_Ac_negg": [0.901081383228302], "r_norm": [1.530577540397644], "g_norm": [0.009232891723513603], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.5639498233795166, "diagnostics_wall_s": 0.06832599639892578, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.5639498233795166, "evaluation_wall_s": 0.024684906005859375}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s0.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s0.json new file mode 100644 index 0000000..e294e81 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6991833448410034}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5, "eval_loss": 0.69349853515625, "wall_s": 2.835977077484131, "val_acc": 0.5, "val_loss": 0.69349853515625, "cos_r_negg": [-0.6283612251281738], "cos_innovation_negg": [-0.6283612251281738], "cos_apical_negg": [-0.6283612251281738], "cos_Ac_negg": [-0.6283612251281738], "r_norm": [0.8063203692436218], "g_norm": [0.0019274819642305374], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.736739158630371, "diagnostics_wall_s": 0.0698397159576416, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.736739158630371, "evaluation_wall_s": 0.024790048599243164}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s1.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s1.json new file mode 100644 index 0000000..1111957 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6932581663131714}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5, "eval_loss": 0.6938357849121094, "wall_s": 2.9175665378570557, "val_acc": 0.5, "val_loss": 0.6938357849121094, "cos_r_negg": [-0.02358771488070488], "cos_innovation_negg": [-0.02358771488070488], "cos_apical_negg": [-0.02358771488070488], "cos_Ac_negg": [-0.02358771488070488], "r_norm": [1.643471360206604], "g_norm": [0.0014360438799485564], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.8199732303619385, "diagnostics_wall_s": 0.06803011894226074, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.8199732303619385, "evaluation_wall_s": 0.024750709533691406}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s2.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s2.json new file mode 100644 index 0000000..aac058f --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6969773173332214}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.4825, "eval_loss": 0.6378177795410156, "wall_s": 2.6224231719970703, "val_acc": 0.4825, "val_loss": 0.6378177795410156, "cos_r_negg": [0.7911241054534912], "cos_innovation_negg": [0.7911241054534912], "cos_apical_negg": [0.7911241054534912], "cos_Ac_negg": [0.7911241054534912], "r_norm": [1.0491777658462524], "g_norm": [0.0050399876199662685], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.497330665588379, "diagnostics_wall_s": 0.09644484519958496, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.497330665588379, "evaluation_wall_s": 0.024004459381103516}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s3.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s3.json new file mode 100644 index 0000000..be6f2dc --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.701652467250824}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5, "eval_loss": 0.6934537048339844, "wall_s": 2.9709041118621826, "val_acc": 0.5, "val_loss": 0.6934537048339844, "cos_r_negg": [-0.6046271324157715], "cos_innovation_negg": [-0.6046271324157715], "cos_apical_negg": [-0.6046271324157715], "cos_Ac_negg": [-0.6046271324157715], "r_norm": [0.6848345994949341], "g_norm": [0.0014354991726577282], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.8740525245666504, "diagnostics_wall_s": 0.06707262992858887, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.8740525245666504, "evaluation_wall_s": 0.024547338485717773}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s4.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s4.json new file mode 100644 index 0000000..e5cc373 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t4_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.8244266510009766}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.706, "eval_loss": 0.5449733581542969, "wall_s": 2.6216139793395996, "val_acc": 0.706, "val_loss": 0.5449733581542969, "cos_r_negg": [0.8991330862045288], "cos_innovation_negg": [0.8991330862045288], "cos_apical_negg": [0.8991330862045288], "cos_Ac_negg": [0.8991330862045288], "r_norm": [1.5177854299545288], "g_norm": [0.009204678237438202], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.516310453414917, "diagnostics_wall_s": 0.07763481140136719, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.516310453414917, "evaluation_wall_s": 0.023111343383789062}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s0.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s0.json new file mode 100644 index 0000000..5e5156d --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7017548084259033}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.579, "eval_loss": 0.6269884643554687, "wall_s": 2.6058199405670166, "val_acc": 0.579, "val_loss": 0.6269884643554687, "cos_r_negg": [0.6208081841468811], "cos_innovation_negg": [0.6208081841468811], "cos_apical_negg": [0.6208081841468811], "cos_Ac_negg": [0.6208081841468811], "r_norm": [0.757482647895813], "g_norm": [0.006873494014143944], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.4861323833465576, "diagnostics_wall_s": 0.09142041206359863, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.4861323833465576, "evaluation_wall_s": 0.023657560348510742}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s1.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s1.json new file mode 100644 index 0000000..4030941 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6929494142532349}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5, "eval_loss": 0.6933207702636719, "wall_s": 2.526723861694336, "val_acc": 0.5, "val_loss": 0.6933207702636719, "cos_r_negg": [0.028608519583940506], "cos_innovation_negg": [0.028608519583940506], "cos_apical_negg": [0.028608519583940506], "cos_Ac_negg": [0.028608519583940506], "r_norm": [1.6420791149139404], "g_norm": [0.0014352130237966776], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.432037353515625, "diagnostics_wall_s": 0.06641173362731934, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.432037353515625, "evaluation_wall_s": 0.023779869079589844}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s2.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s2.json new file mode 100644 index 0000000..e080bea --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6991405487060547}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.585, "eval_loss": 0.6117306213378906, "wall_s": 2.6806247234344482, "val_acc": 0.585, "val_loss": 0.6117306213378906, "cos_r_negg": [0.7898202538490295], "cos_innovation_negg": [0.7898202538490295], "cos_apical_negg": [0.7898202538490295], "cos_Ac_negg": [0.7898202538490295], "r_norm": [1.0321531295776367], "g_norm": [0.0070411646738648415], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.5764365196228027, "diagnostics_wall_s": 0.06628870964050293, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.5764365196228027, "evaluation_wall_s": 0.0332484245300293}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s3.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s3.json new file mode 100644 index 0000000..a8dbb9f --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7045606374740601}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.531, "eval_loss": 0.6795563354492188, "wall_s": 2.754044532775879, "val_acc": 0.531, "val_loss": 0.6795563354492188, "cos_r_negg": [0.43447092175483704], "cos_innovation_negg": [0.43447092175483704], "cos_apical_negg": [0.43447092175483704], "cos_Ac_negg": [0.43447092175483704], "r_norm": [0.6758479475975037], "g_norm": [0.003239097073674202], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.646357774734497, "diagnostics_wall_s": 0.0718235969543457, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.646357774734497, "evaluation_wall_s": 0.03132176399230957}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s4.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s4.json new file mode 100644 index 0000000..cdb9ff7 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d1_t5_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7986906170845032}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.729, "eval_loss": 0.5289509582519532, "wall_s": 2.5730667114257812, "val_acc": 0.729, "val_loss": 0.5289509582519532, "cos_r_negg": [0.9014049172401428], "cos_innovation_negg": [0.9014049172401428], "cos_apical_negg": [0.9014049172401428], "cos_Ac_negg": [0.9014049172401428], "r_norm": [1.530706524848938], "g_norm": [0.00917032826691866], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.4784669876098633, "diagnostics_wall_s": 0.06583094596862793, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.4784669876098633, "evaluation_wall_s": 0.024263620376586914}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396224, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s0.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s0.json new file mode 100644 index 0000000..c69467b --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6894387006759644}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9795, "eval_loss": 0.15662760162353515, "wall_s": 5.921708583831787, "val_acc": 0.9795, "val_loss": 0.15662760162353515, "cos_r_negg": [0.9521638751029968, 0.8179508447647095, 0.6949918270111084, 0.4908371865749359], "cos_innovation_negg": [0.9521638751029968, 0.8179508447647095, 0.6949918270111084, 0.4908371865749359], "cos_apical_negg": [0.9521638751029968, 0.8179508447647095, 0.6949918270111084, 0.4908371865749359], "cos_Ac_negg": [0.9521638751029968, 0.8179508447647095, 0.6949918270111084, 0.4908371865749359], "r_norm": [0.41740578413009644, 0.31533753871917725, 0.27202141284942627, 0.36710071563720703], "g_norm": [0.01194473635405302, 0.009706087410449982, 0.006237569730728865, 0.004933253861963749], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.457855761051178, 0.6056836247444153, 0.42167747020721436], "lesion_eval_acc": 0.6255, "lesion_eval_loss": 0.5102483673095703, "lesion_acc_drop": 0.3540000000000001}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.790801286697388, "diagnostics_wall_s": 0.09026122093200684, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.790801286697388, "evaluation_wall_s": 0.03596091270446777}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s1.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s1.json new file mode 100644 index 0000000..65ecc8b --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.745452880859375}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.543, "eval_loss": 0.6392783203125, "wall_s": 5.489480972290039, "val_acc": 0.543, "val_loss": 0.6392783203125, "cos_r_negg": [0.2431916743516922, -0.5103000402450562, -0.5922247767448425, 0.6179416179656982], "cos_innovation_negg": [0.2431916743516922, -0.5103000402450562, -0.5922247767448425, 0.6179416179656982], "cos_apical_negg": [0.2431916743516922, -0.5103000402450562, -0.5922247767448425, 0.6179416179656982], "cos_Ac_negg": [0.2431916743516922, -0.5103000402450562, -0.5922247767448425, 0.6179416179656982], "r_norm": [1.7762105464935303, 1.0314278602600098, 1.8215694427490234, 1.0765464305877686], "g_norm": [0.008118749596178532, 0.007465130649507046, 0.007426619064062834, 0.007361199241131544], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.1930406391620636, 0.5746530294418335, 0.8953241109848022], "lesion_eval_acc": 0.4435, "lesion_eval_loss": 0.9390199584960938, "lesion_acc_drop": 0.09950000000000003}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.3617517948150635, "diagnostics_wall_s": 0.0872492790222168, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.3617517948150635, "evaluation_wall_s": 0.0360102653503418}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s2.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s2.json new file mode 100644 index 0000000..378dbe9 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6903389096260071}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.562, "eval_loss": 0.589281005859375, "wall_s": 5.816173076629639, "val_acc": 0.562, "val_loss": 0.589281005859375, "cos_r_negg": [0.8700941801071167, -0.294040322303772, 0.43819576501846313, 0.7383877038955688], "cos_innovation_negg": [0.8700941801071167, -0.294040322303772, 0.43819576501846313, 0.7383877038955688], "cos_apical_negg": [0.8700941801071167, -0.294040322303772, 0.43819576501846313, 0.7383877038955688], "cos_Ac_negg": [0.8700941801071167, -0.294040322303772, 0.43819576501846313, 0.7383877038955688], "r_norm": [0.8321583271026611, 1.4200289249420166, 0.9375735521316528, 1.002671241760254], "g_norm": [0.011334361508488655, 0.01215396262705326, 0.009067073464393616, 0.006586203817278147], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.6343659162521362, 0.7832711935043335, 1.1649914979934692], "lesion_eval_acc": 0.5, "lesion_eval_loss": 0.9280475463867187, "lesion_acc_drop": 0.062000000000000055}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.685602903366089, "diagnostics_wall_s": 0.08889198303222656, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.685602903366089, "evaluation_wall_s": 0.03661346435546875}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s3.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s3.json new file mode 100644 index 0000000..2d6f2b2 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7031912803649902}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.6715, "eval_loss": 0.4627419128417969, "wall_s": 5.94448184967041, "val_acc": 0.6715, "val_loss": 0.4627419128417969, "cos_r_negg": [0.2979603111743927, 0.234616219997406, 0.9765276908874512, 0.8881837725639343], "cos_innovation_negg": [0.2979603111743927, 0.234616219997406, 0.9765276908874512, 0.8881837725639343], "cos_apical_negg": [0.2979603111743927, 0.234616219997406, 0.9765276908874512, 0.8881837725639343], "cos_Ac_negg": [0.2979603111743927, 0.234616219997406, 0.9765276908874512, 0.8881837725639343], "r_norm": [0.9817248582839966, 1.003795862197876, 1.373692274093628, 0.7375629544258118], "g_norm": [0.02128533646464348, 0.02099933847784996, 0.018753932788968086, 0.008710246533155441], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.8674094080924988, 0.2306327074766159, 0.5033317804336548], "lesion_eval_acc": 0.572, "lesion_eval_loss": 1.0473679809570313, "lesion_acc_drop": 0.09950000000000003}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.806238412857056, "diagnostics_wall_s": 0.09033513069152832, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.806238412857056, "evaluation_wall_s": 0.04309201240539551}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s4.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s4.json new file mode 100644 index 0000000..646d5f9 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t3_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 1.1787136793136597}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.7685, "eval_loss": 0.45205384826660155, "wall_s": 5.704097032546997, "val_acc": 0.7685, "val_loss": 0.45205384826660155, "cos_r_negg": [0.4741443395614624, 0.49037081003189087, 0.7511203289031982, 0.9619051814079285], "cos_innovation_negg": [0.4741443395614624, 0.49037081003189087, 0.7511203289031982, 0.9619051814079285], "cos_apical_negg": [0.4741443395614624, 0.49037081003189087, 0.7511203289031982, 0.9619051814079285], "cos_Ac_negg": [0.4741443395614624, 0.49037081003189087, 0.7511203289031982, 0.9619051814079285], "r_norm": [0.9297653436660767, 0.9775343537330627, 1.108774185180664, 1.5429840087890625], "g_norm": [0.013986989855766296, 0.01210818998515606, 0.011616159230470657, 0.007869511842727661], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.47147539258003235, 0.24181345105171204, 0.2206602841615677], "lesion_eval_acc": 0.5, "lesion_eval_loss": 1.1133299560546874, "lesion_acc_drop": 0.26849999999999996}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.555422306060791, "diagnostics_wall_s": 0.09232354164123535, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.555422306060791, "evaluation_wall_s": 0.04271078109741211}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s0.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s0.json new file mode 100644 index 0000000..6273e0c --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6999248266220093}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9985, "eval_loss": 0.22465267181396484, "wall_s": 5.6400933265686035, "val_acc": 0.9985, "val_loss": 0.22465267181396484, "cos_r_negg": [0.942209780216217, 0.8142115473747253, 0.8885233402252197, 0.4583420753479004], "cos_innovation_negg": [0.942209780216217, 0.8142115473747253, 0.8885233402252197, 0.4583420753479004], "cos_apical_negg": [0.942209780216217, 0.8142115473747253, 0.8885233402252197, 0.4583420753479004], "cos_Ac_negg": [0.942209780216217, 0.8142115473747253, 0.8885233402252197, 0.4583420753479004], "r_norm": [0.5966967940330505, 0.4507865309715271, 0.38886457681655884, 0.5247838497161865], "g_norm": [0.012931514531373978, 0.012033550068736076, 0.008448587730526924, 0.006324116140604019], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.406389057636261, 0.9502606391906738, 0.5212638974189758], "lesion_eval_acc": 0.5, "lesion_eval_loss": 3.1309146728515627, "lesion_acc_drop": 0.49850000000000005}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.509395122528076, "diagnostics_wall_s": 0.09018397331237793, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.509395122528076, "evaluation_wall_s": 0.03590655326843262}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s1.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s1.json new file mode 100644 index 0000000..009a4a0 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7811087965965271}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.685, "eval_loss": 0.6601065063476562, "wall_s": 5.832489252090454, "val_acc": 0.685, "val_loss": 0.6601065063476562, "cos_r_negg": [0.07269817590713501, 0.8837659358978271, 0.8391057252883911, 0.4753904342651367], "cos_innovation_negg": [0.07269817590713501, 0.8837659358978271, 0.8391057252883911, 0.4753904342651367], "cos_apical_negg": [0.07269817590713501, 0.8837659358978271, 0.8391057252883911, 0.4753904342651367], "cos_Ac_negg": [0.07269817590713501, 0.8837659358978271, 0.8391057252883911, 0.4753904342651367], "r_norm": [1.7132225036621094, 0.9948513507843018, 1.7569729089736938, 1.0383697748184204], "g_norm": [0.014306677505373955, 0.01710614748299122, 0.009541677311062813, 0.00657348707318306], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.3420197069644928, 0.5112898945808411, 0.7219170928001404], "lesion_eval_acc": 0.5, "lesion_eval_loss": 1.06602490234375, "lesion_acc_drop": 0.18500000000000005}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.699199676513672, "diagnostics_wall_s": 0.09122276306152344, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.699199676513672, "evaluation_wall_s": 0.0374603271484375}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s2.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s2.json new file mode 100644 index 0000000..3619178 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6940909624099731}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5485, "eval_loss": 0.5851892395019531, "wall_s": 5.777306079864502, "val_acc": 0.5485, "val_loss": 0.5851892395019531, "cos_r_negg": [0.9218685626983643, -0.5135247111320496, 0.22491347789764404, 0.8067145347595215], "cos_innovation_negg": [0.9218685626983643, -0.5135247111320496, 0.22491347789764404, 0.8067145347595215], "cos_apical_negg": [0.9218685626983643, -0.5135247111320496, 0.22491347789764404, 0.8067145347595215], "cos_Ac_negg": [0.9218685626983643, -0.5135247111320496, 0.22491347789764404, 0.8067145347595215], "r_norm": [0.8557249307632446, 1.4602439403533936, 0.964125394821167, 1.031066656112671], "g_norm": [0.011888370849192142, 0.014062203466892242, 0.01147821731865406, 0.007162131369113922], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.6014787554740906, 0.65707927942276, 1.127349853515625], "lesion_eval_acc": 0.375, "lesion_eval_loss": 0.8358504028320313, "lesion_acc_drop": 0.1735}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.63305401802063, "diagnostics_wall_s": 0.0931241512298584, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.63305401802063, "evaluation_wall_s": 0.0462956428527832}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s3.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s3.json new file mode 100644 index 0000000..f03fbeb --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6956391334533691}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9885, "eval_loss": 0.24751303100585936, "wall_s": 5.7301390171051025, "val_acc": 0.9885, "val_loss": 0.24751303100585936, "cos_r_negg": [0.10184620320796967, -0.21835723519325256, 0.8730567693710327, 0.8644888401031494], "cos_innovation_negg": [0.10184620320796967, -0.21835723519325256, 0.8730567693710327, 0.8644888401031494], "cos_apical_negg": [0.10184620320796967, -0.21835723519325256, 0.8730567693710327, 0.8644888401031494], "cos_Ac_negg": [0.10184620320796967, -0.21835723519325256, 0.8730567693710327, 0.8644888401031494], "r_norm": [0.5777345299720764, 0.5907230377197266, 0.8084032535552979, 0.4340478777885437], "g_norm": [0.022050045430660248, 0.017077725380659103, 0.014984814450144768, 0.006239097099751234], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.7263292074203491, 0.2723751962184906, 0.4880276322364807], "lesion_eval_acc": 0.563, "lesion_eval_loss": 0.9203932800292969, "lesion_acc_drop": 0.4255000000000001}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.590011119842529, "diagnostics_wall_s": 0.09214162826538086, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.590011119842529, "evaluation_wall_s": 0.0432126522064209}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s4.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s4.json new file mode 100644 index 0000000..227fe3b --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t4_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 1.0611846446990967}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.664, "eval_loss": 0.5397358093261718, "wall_s": 6.022627115249634, "val_acc": 0.664, "val_loss": 0.5397358093261718, "cos_r_negg": [0.3534785807132721, 0.05067330598831177, -0.29257455468177795, 0.8511010408401489], "cos_innovation_negg": [0.3534785807132721, 0.05067330598831177, -0.29257455468177795, 0.8511010408401489], "cos_apical_negg": [0.3534785807132721, 0.05067330598831177, -0.29257455468177795, 0.8511010408401489], "cos_Ac_negg": [0.3534785807132721, 0.05067330598831177, -0.29257455468177795, 0.8511010408401489], "r_norm": [1.1264357566833496, 1.1843092441558838, 1.3433096408843994, 1.8693666458129883], "g_norm": [0.009046060033142567, 0.007576550357043743, 0.006103368941694498, 0.00424253661185503], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.46134325861930847, 0.3677596151828766, 0.10889279097318649], "lesion_eval_acc": 0.631, "lesion_eval_loss": 0.7211289978027344, "lesion_acc_drop": 0.03300000000000003}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.873251914978027, "diagnostics_wall_s": 0.1019449234008789, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.873251914978027, "evaluation_wall_s": 0.04246401786804199}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s0.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s0.json new file mode 100644 index 0000000..1bcc6e4 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6962454319000244}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9815, "eval_loss": 0.22388123321533204, "wall_s": 5.65465235710144, "val_acc": 0.9815, "val_loss": 0.22388123321533204, "cos_r_negg": [0.9438206553459167, 0.8032230138778687, 0.8777164220809937, 0.43844056129455566], "cos_innovation_negg": [0.9438206553459167, 0.8032230138778687, 0.8777164220809937, 0.43844056129455566], "cos_apical_negg": [0.9438206553459167, 0.8032230138778687, 0.8777164220809937, 0.43844056129455566], "cos_Ac_negg": [0.9438206553459167, 0.8032230138778687, 0.8777164220809937, 0.43844056129455566], "r_norm": [0.5938228964805603, 0.44861534237861633, 0.3869916498661041, 0.5222563147544861], "g_norm": [0.012756235897541046, 0.011698182672262192, 0.008216079324483871, 0.006365514826029539], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.4115850031375885, 0.9452176690101624, 0.536113440990448], "lesion_eval_acc": 0.5, "lesion_eval_loss": 3.2792166748046876, "lesion_acc_drop": 0.48150000000000004}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.5172882080078125, "diagnostics_wall_s": 0.09975981712341309, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.5172882080078125, "evaluation_wall_s": 0.032903194427490234}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s1.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s1.json new file mode 100644 index 0000000..a2acb98 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.761730432510376}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.642, "eval_loss": 0.6766459045410156, "wall_s": 5.703474283218384, "val_acc": 0.642, "val_loss": 0.6766459045410156, "cos_r_negg": [0.05444283410906792, 1.0, -0.7967917919158936, 0.46257951855659485], "cos_innovation_negg": [0.05444283410906792, 1.0, -0.7967917919158936, 0.46257951855659485], "cos_apical_negg": [0.05444283410906792, 1.0, -0.7967917919158936, 0.46257951855659485], "cos_Ac_negg": [0.05444283410906792, 1.0, -0.7967917919158936, 0.46257951855659485], "r_norm": [1.8942536115646362, 1.099974274635315, 1.942626953125, 1.148091197013855], "g_norm": [0.0035502701066434383, 0.0035288487561047077, 0.0034656068310141563, 0.00394304096698761], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.17661087214946747, 0.5922853350639343, 0.8051239848136902], "lesion_eval_acc": 0.484, "lesion_eval_loss": 0.7626482849121093, "lesion_acc_drop": 0.15800000000000003}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.565644264221191, "diagnostics_wall_s": 0.10085153579711914, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.565644264221191, "evaluation_wall_s": 0.032243967056274414}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s2.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s2.json new file mode 100644 index 0000000..eb08782 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6929397583007812}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.6995, "eval_loss": 0.5813926086425781, "wall_s": 5.845289707183838, "val_acc": 0.6995, "val_loss": 0.5813926086425781, "cos_r_negg": [0.8734152913093567, -0.28353220224380493, 0.42803987860679626, 0.7424302101135254], "cos_innovation_negg": [0.8734152913093567, -0.28353220224380493, 0.42803987860679626, 0.7424302101135254], "cos_apical_negg": [0.8734152913093567, -0.28353220224380493, 0.42803987860679626, 0.7424302101135254], "cos_Ac_negg": [0.8734152913093567, -0.28353220224380493, 0.42803987860679626, 0.7424302101135254], "r_norm": [0.8465811610221863, 1.4446406364440918, 0.9538233280181885, 1.0200493335723877], "g_norm": [0.012448800727725029, 0.013021832332015038, 0.009601039811968803, 0.006786726880818605], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.6423255205154419, 0.7898757457733154, 1.1479154825210571], "lesion_eval_acc": 0.5, "lesion_eval_loss": 0.8362770080566406, "lesion_acc_drop": 0.1995}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.716195583343506, "diagnostics_wall_s": 0.08849334716796875, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.716195583343506, "evaluation_wall_s": 0.03606820106506348}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s3.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s3.json new file mode 100644 index 0000000..5ff9b4b --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.697464108467102}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9945, "eval_loss": 0.16081343841552734, "wall_s": 8.113498210906982, "val_acc": 0.9945, "val_loss": 0.16081343841552734, "cos_r_negg": [0.026243962347507477, 0.2266158163547516, 0.8410705924034119, 0.8464125394821167], "cos_innovation_negg": [0.026243962347507477, 0.2266158163547516, 0.8410705924034119, 0.8464125394821167], "cos_apical_negg": [0.026243962347507477, 0.2266158163547516, 0.8410705924034119, 0.8464125394821167], "cos_Ac_negg": [0.026243962347507477, 0.2266158163547516, 0.8410705924034119, 0.8464125394821167], "r_norm": [0.3890058398246765, 0.3977513611316681, 0.5443218350410461, 0.29225730895996094], "g_norm": [0.021645480766892433, 0.01600879430770874, 0.012230262160301208, 0.005071072839200497], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.5650879144668579, 0.29304924607276917, 0.4338463842868805], "lesion_eval_acc": 0.5, "lesion_eval_loss": 1.4438804931640625, "lesion_acc_drop": 0.49450000000000005}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 7.970989465713501, "diagnostics_wall_s": 0.09794330596923828, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 7.970989465713501, "evaluation_wall_s": 0.03922462463378906}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s4.json b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s4.json new file mode 100644 index 0000000..fe4e672 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "dfa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_dfa_d4_t5_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 1.0589196681976318}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.838, "eval_loss": 0.4642987823486328, "wall_s": 6.236632823944092, "val_acc": 0.838, "val_loss": 0.4642987823486328, "cos_r_negg": [0.40154650807380676, 0.3095412850379944, -0.31228357553482056, 0.903065025806427], "cos_innovation_negg": [0.40154650807380676, 0.3095412850379944, -0.31228357553482056, 0.903065025806427], "cos_apical_negg": [0.40154650807380676, 0.3095412850379944, -0.31228357553482056, 0.903065025806427], "cos_Ac_negg": [0.40154650807380676, 0.3095412850379944, -0.31228357553482056, 0.903065025806427], "r_norm": [0.9127594232559204, 0.959654688835144, 1.088494062423706, 1.5147619247436523], "g_norm": [0.018689187243580818, 0.016177576035261154, 0.013912013731896877, 0.00691058486700058], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.7693451642990112, 0.5721911191940308, 0.10891760885715485], "lesion_eval_acc": 0.5615, "lesion_eval_loss": 1.4469329223632812, "lesion_acc_drop": 0.27649999999999997}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 6.087916851043701, "diagnostics_wall_s": 0.09988141059875488, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 6.087916851043701, "evaluation_wall_s": 0.04389476776123047}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17558016, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s0.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s0.json new file mode 100644 index 0000000..381253c --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t3_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7108339071273804}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.496, "eval_loss": 0.6309637756347656, "wall_s": 3.2597343921661377, "val_acc": 0.496, "val_loss": 0.6309637756347656, "cos_fa_negg": [0.6865530014038086], "loss": 0.6230919361114502}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 3.1604232788085938, "diagnostics_wall_s": 0.06451559066772461, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 3.1604232788085938, "evaluation_wall_s": 0.029317855834960938}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s1.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s1.json new file mode 100644 index 0000000..742a0a2 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t3_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6921945810317993}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.519, "eval_loss": 0.6449997863769531, "wall_s": 2.751523971557617, "val_acc": 0.519, "val_loss": 0.6449997863769531, "cos_fa_negg": [0.7812427282333374], "loss": 0.6396073698997498}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.679589033126831, "diagnostics_wall_s": 0.044667720794677734, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.679589033126831, "evaluation_wall_s": 0.022849082946777344}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s2.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s2.json new file mode 100644 index 0000000..7bd3212 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t3_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7062801718711853}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.4985, "eval_loss": 0.6328848266601562, "wall_s": 2.706055164337158, "val_acc": 0.4985, "val_loss": 0.6328848266601562, "cos_fa_negg": [0.9326016306877136], "loss": 0.6252561211585999}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.6197903156280518, "diagnostics_wall_s": 0.05295825004577637, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.6197903156280518, "evaluation_wall_s": 0.028357267379760742}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s3.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s3.json new file mode 100644 index 0000000..4c65d44 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t3_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7154970169067383}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5085, "eval_loss": 0.6416968078613281, "wall_s": 2.8162930011749268, "val_acc": 0.5085, "val_loss": 0.6416968078613281, "cos_fa_negg": [0.8984099626541138], "loss": 0.6353766918182373}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.740715503692627, "diagnostics_wall_s": 0.04680061340332031, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.740715503692627, "evaluation_wall_s": 0.023898601531982422}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s4.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s4.json new file mode 100644 index 0000000..96dcdda --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t3_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t3_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7816296815872192}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.494, "eval_loss": 0.6338832092285156, "wall_s": 2.6501924991607666, "val_acc": 0.494, "val_loss": 0.6338832092285156, "cos_fa_negg": [0.8827385306358337], "loss": 0.6257074475288391}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.577127695083618, "diagnostics_wall_s": 0.04536128044128418, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.577127695083618, "evaluation_wall_s": 0.022895336151123047}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s0.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s0.json new file mode 100644 index 0000000..bee90d7 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t4_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6991833448410034}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.483, "eval_loss": 0.6245528259277344, "wall_s": 2.771667718887329, "val_acc": 0.483, "val_loss": 0.6245528259277344, "cos_fa_negg": [0.6836392283439636], "loss": 0.6318127512931824}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.696347713470459, "diagnostics_wall_s": 0.04625844955444336, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.696347713470459, "evaluation_wall_s": 0.024511098861694336}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s1.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s1.json new file mode 100644 index 0000000..5acab8e --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t4_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6932582855224609}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.4835, "eval_loss": 0.6447336120605469, "wall_s": 2.617647171020508, "val_acc": 0.4835, "val_loss": 0.6447336120605469, "cos_fa_negg": [0.6708382964134216], "loss": 0.6494970321655273}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.537152051925659, "diagnostics_wall_s": 0.04667353630065918, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.537152051925659, "evaluation_wall_s": 0.02427506446838379}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s2.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s2.json new file mode 100644 index 0000000..a1ba0f9 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t4_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6969773173332214}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.495, "eval_loss": 0.6357405395507812, "wall_s": 2.612156867980957, "val_acc": 0.495, "val_loss": 0.6357405395507812, "cos_fa_negg": [0.9323561191558838], "loss": 0.6416018009185791}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.526963472366333, "diagnostics_wall_s": 0.045078277587890625, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.526963472366333, "evaluation_wall_s": 0.024465084075927734}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s3.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s3.json new file mode 100644 index 0000000..bba5695 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t4_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7016525268554688}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.4915, "eval_loss": 0.6374026794433594, "wall_s": 2.586143970489502, "val_acc": 0.4915, "val_loss": 0.6374026794433594, "cos_fa_negg": [0.8959587812423706], "loss": 0.6429191827774048}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.5031251907348633, "diagnostics_wall_s": 0.04419064521789551, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.5031251907348633, "evaluation_wall_s": 0.024229764938354492}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s4.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s4.json new file mode 100644 index 0000000..34bc11e --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t4_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t4_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.8244265913963318}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.492, "eval_loss": 0.626970458984375, "wall_s": 3.537045955657959, "val_acc": 0.492, "val_loss": 0.626970458984375, "cos_fa_negg": [0.8812520503997803], "loss": 0.6338187456130981}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 3.462566375732422, "diagnostics_wall_s": 0.04522585868835449, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 3.462566375732422, "evaluation_wall_s": 0.024308443069458008}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s0.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s0.json new file mode 100644 index 0000000..39e98af --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t5_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7017548680305481}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.584, "eval_loss": 0.6172794189453125, "wall_s": 2.7091896533966064, "val_acc": 0.584, "val_loss": 0.6172794189453125, "cos_fa_negg": [0.6783945560455322], "loss": 0.6389129161834717}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.6296253204345703, "diagnostics_wall_s": 0.04638957977294922, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.6296253204345703, "evaluation_wall_s": 0.02850627899169922}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s1.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s1.json new file mode 100644 index 0000000..bdc77ba --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t5_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6929495334625244}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.586, "eval_loss": 0.6323392028808594, "wall_s": 2.875108242034912, "val_acc": 0.586, "val_loss": 0.6323392028808594, "cos_fa_negg": [0.7839969396591187], "loss": 0.6496721506118774}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.769113540649414, "diagnostics_wall_s": 0.07669377326965332, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.769113540649414, "evaluation_wall_s": 0.024336814880371094}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s2.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s2.json new file mode 100644 index 0000000..d09b06d --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t5_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6991406083106995}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5885, "eval_loss": 0.6194549560546875, "wall_s": 2.663761615753174, "val_acc": 0.5885, "val_loss": 0.6194549560546875, "cos_fa_negg": [0.9312798976898193], "loss": 0.6403682231903076}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.5766654014587402, "diagnostics_wall_s": 0.058121681213378906, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.5766654014587402, "evaluation_wall_s": 0.024434566497802734}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s3.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s3.json new file mode 100644 index 0000000..4444ecf --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t5_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7045606970787048}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.595, "eval_loss": 0.6301483154296875, "wall_s": 2.8391122817993164, "val_acc": 0.595, "val_loss": 0.6301483154296875, "cos_fa_negg": [0.8990697264671326], "loss": 0.6480233669281006}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.7467451095581055, "diagnostics_wall_s": 0.06417083740234375, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.7467451095581055, "evaluation_wall_s": 0.023821592330932617}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s4.json b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s4.json new file mode 100644 index 0000000..14aaa23 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d1_t5_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d1_t5_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.798690676689148}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5815, "eval_loss": 0.6243822631835938, "wall_s": 2.6881442070007324, "val_acc": 0.5815, "val_loss": 0.6243822631835938, "cos_fa_negg": [0.8764917254447937], "loss": 0.6441296339035034}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 2.6095056533813477, "diagnostics_wall_s": 0.046404361724853516, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 2.6095056533813477, "evaluation_wall_s": 0.027692794799804688}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17310208, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s0.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s0.json new file mode 100644 index 0000000..1f84c36 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t3_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6894387006759644}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.643, "eval_loss": 0.45172303771972655, "wall_s": 5.977975368499756, "val_acc": 0.643, "val_loss": 0.45172303771972655, "cos_fa_negg": [0.2638549208641052, 0.9719889163970947, 0.7757389545440674, 0.8117539286613464], "loss": 0.46842044591903687, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.7161180973052979, 0.6727262735366821, 0.4412742555141449], "lesion_eval_acc": 0.6865, "lesion_eval_loss": 3.85023046875, "lesion_acc_drop": -0.04349999999999998}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.851597785949707, "diagnostics_wall_s": 0.06371569633483887, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.851597785949707, "evaluation_wall_s": 0.0580143928527832}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s1.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s1.json new file mode 100644 index 0000000..a6b6b54 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t3_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7454528212547302}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.4135, "eval_loss": 0.6887763366699219, "wall_s": 5.930422306060791, "val_acc": 0.4135, "val_loss": 0.6887763366699219, "cos_fa_negg": [0.04215267300605774, -0.44500279426574707, 0.6012042760848999, 0.5575250387191772], "loss": 0.6869109272956848, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.08905722945928574, 0.6625007390975952, 0.4995851516723633], "lesion_eval_acc": 0.5, "lesion_eval_loss": 0.7449973754882813, "lesion_acc_drop": -0.08650000000000002}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.805293560028076, "diagnostics_wall_s": 0.06297588348388672, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.805293560028076, "evaluation_wall_s": 0.05720639228820801}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s2.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s2.json new file mode 100644 index 0000000..23714d2 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t3_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6903390288352966}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9605, "eval_loss": 0.3230642547607422, "wall_s": 6.036694049835205, "val_acc": 0.9605, "val_loss": 0.3230642547607422, "cos_fa_negg": [0.7445300221443176, 0.7388441562652588, 0.4545643925666809, 0.8355056047439575], "loss": 0.33325669169425964, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.4006965458393097, 0.36283430457115173, 0.6814932823181152], "lesion_eval_acc": 0.5755, "lesion_eval_loss": 2.8021234130859374, "lesion_acc_drop": 0.385}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.89646053314209, "diagnostics_wall_s": 0.0650641918182373, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.89646053314209, "evaluation_wall_s": 0.05990338325500488}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s3.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s3.json new file mode 100644 index 0000000..7f4631e --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t3_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7031912207603455}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9615, "eval_loss": 0.12419326782226563, "wall_s": 5.964731454849243, "val_acc": 0.9615, "val_loss": 0.12419326782226563, "cos_fa_negg": [0.775720477104187, 0.9548989534378052, 0.9402242302894592, 0.9714613556861877], "loss": 0.1315620094537735, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.6108824610710144, 0.4098055958747864, 0.4748568534851074], "lesion_eval_acc": 0.722, "lesion_eval_loss": 0.4521162414550781, "lesion_acc_drop": 0.23950000000000005}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.821832656860352, "diagnostics_wall_s": 0.07980728149414062, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.821832656860352, "evaluation_wall_s": 0.05821967124938965}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s4.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s4.json new file mode 100644 index 0000000..95086b6 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t3_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t3_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 1.1787136793136597}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.931, "eval_loss": 0.3203551635742187, "wall_s": 6.011535882949829, "val_acc": 0.931, "val_loss": 0.3203551635742187, "cos_fa_negg": [0.6590604186058044, 0.6063621640205383, 0.9661731719970703, 0.9383794069290161], "loss": 0.331265926361084, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.5787550210952759, 0.22193826735019684, 0.1889389008283615], "lesion_eval_acc": 0.5945, "lesion_eval_loss": 1.7418964233398437, "lesion_acc_drop": 0.3365}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.88528299331665, "diagnostics_wall_s": 0.06427359580993652, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.88528299331665, "evaluation_wall_s": 0.05738115310668945}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s0.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s0.json new file mode 100644 index 0000000..a5aed36 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t4_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6999248266220093}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.744, "eval_loss": 0.42777337646484376, "wall_s": 5.5505547523498535, "val_acc": 0.744, "val_loss": 0.42777337646484376, "cos_fa_negg": [0.47339576482772827, 0.9371519088745117, 0.7827718257904053, 0.8872032761573792], "loss": 0.44848498702049255, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.8459184765815735, 0.657263457775116, 0.42491817474365234], "lesion_eval_acc": 0.724, "lesion_eval_loss": 3.0639716796875, "lesion_acc_drop": 0.020000000000000018}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.423597812652588, "diagnostics_wall_s": 0.06495881080627441, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.423597812652588, "evaluation_wall_s": 0.05747795104980469}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s1.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s1.json new file mode 100644 index 0000000..0b94fc0 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t4_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7811087369918823}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.4, "eval_loss": 0.6864822082519532, "wall_s": 6.778041124343872, "val_acc": 0.4, "val_loss": 0.6864822082519532, "cos_fa_negg": [0.02073442004621029, -1.0, 0.6035793423652649, 0.5548111796379089], "loss": 0.6864082217216492, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.06909126788377762, 0.6824622750282288, 0.4917472302913666], "lesion_eval_acc": 0.5, "lesion_eval_loss": 0.7324013671875, "lesion_acc_drop": -0.09999999999999998}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 6.645745277404785, "diagnostics_wall_s": 0.0672597885131836, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 6.645745277404785, "evaluation_wall_s": 0.05965900421142578}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s2.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s2.json new file mode 100644 index 0000000..46ee296 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t4_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6940910816192627}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9625, "eval_loss": 0.3486024932861328, "wall_s": 5.752170085906982, "val_acc": 0.9625, "val_loss": 0.3486024932861328, "cos_fa_negg": [0.6445547938346863, 0.7063145637512207, 0.5975035429000854, 0.6468939185142517], "loss": 0.3593002259731293, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.4394884407520294, 0.4684455692768097, 0.7101873755455017], "lesion_eval_acc": 0.593, "lesion_eval_loss": 4.30203076171875, "lesion_acc_drop": 0.36950000000000005}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.619373559951782, "diagnostics_wall_s": 0.06777167320251465, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.619373559951782, "evaluation_wall_s": 0.060227155685424805}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s3.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s3.json new file mode 100644 index 0000000..29b0e50 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t4_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6956391334533691}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9905, "eval_loss": 0.1422619934082031, "wall_s": 6.533494234085083, "val_acc": 0.9905, "val_loss": 0.1422619934082031, "cos_fa_negg": [0.9072752594947815, 0.9884588718414307, -0.03423362225294113, 0.884361982345581], "loss": 0.1397797167301178, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.4115552604198456, 0.37038931250572205, 0.5144444704055786], "lesion_eval_acc": 0.6105, "lesion_eval_loss": 0.726352294921875, "lesion_acc_drop": 0.38}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 6.397737979888916, "diagnostics_wall_s": 0.06429815292358398, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 6.397737979888916, "evaluation_wall_s": 0.0658116340637207}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s4.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s4.json new file mode 100644 index 0000000..d693cd1 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t4_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t4_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 1.0611846446990967}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9785, "eval_loss": 0.251977668762207, "wall_s": 6.115600824356079, "val_acc": 0.9785, "val_loss": 0.251977668762207, "cos_fa_negg": [0.6830018758773804, 0.5683583617210388, 0.877593457698822, 0.9262003302574158], "loss": 0.2565094530582428, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.515380322933197, 0.24766726791858673, 0.20463277399539948], "lesion_eval_acc": 0.5, "lesion_eval_loss": 1.5873077392578125, "lesion_acc_drop": 0.47850000000000004}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.984712362289429, "diagnostics_wall_s": 0.06232404708862305, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.984712362289429, "evaluation_wall_s": 0.06357192993164062}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s0.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s0.json new file mode 100644 index 0000000..ecaf4c0 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t5_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6962454319000244}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.83, "eval_loss": 0.4593400115966797, "wall_s": 5.907103061676025, "val_acc": 0.83, "val_loss": 0.4593400115966797, "cos_fa_negg": [0.27096259593963623, 0.97391676902771, 0.7575252056121826, 0.8225903511047363], "loss": 0.48334550857543945, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.7226333022117615, 0.6850886344909668, 0.4555051624774933], "lesion_eval_acc": 0.678, "lesion_eval_loss": 3.8207205810546876, "lesion_acc_drop": 0.1519999999999999}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.782327890396118, "diagnostics_wall_s": 0.06463479995727539, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.782327890396118, "evaluation_wall_s": 0.05527234077453613}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s1.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s1.json new file mode 100644 index 0000000..95453d8 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t5_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7617303729057312}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.642, "eval_loss": 0.6828298950195313, "wall_s": 5.8658154010772705, "val_acc": 0.642, "val_loss": 0.6828298950195313, "cos_fa_negg": [0.10061662644147873, -0.3576048016548157, 0.617953896522522, 0.5752323865890503], "loss": 0.687859833240509, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.08367646485567093, 0.6698669195175171, 0.5135611891746521], "lesion_eval_acc": 0.5, "lesion_eval_loss": 0.7318787231445313, "lesion_acc_drop": 0.14200000000000002}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.732851028442383, "diagnostics_wall_s": 0.06332921981811523, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.732851028442383, "evaluation_wall_s": 0.05716848373413086}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s2.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s2.json new file mode 100644 index 0000000..aaeb7e9 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t5_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.692939817905426}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9625, "eval_loss": 0.2898576354980469, "wall_s": 5.517089366912842, "val_acc": 0.9625, "val_loss": 0.2898576354980469, "cos_fa_negg": [0.7879692912101746, 0.913291335105896, 0.7581121921539307, 0.8298001289367676], "loss": 0.3003949522972107, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.24749979376792908, 0.5770268440246582, 0.630447268486023], "lesion_eval_acc": 0.5, "lesion_eval_loss": 2.861123779296875, "lesion_acc_drop": 0.4625}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.391567707061768, "diagnostics_wall_s": 0.06524777412414551, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.391567707061768, "evaluation_wall_s": 0.055678606033325195}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s3.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s3.json new file mode 100644 index 0000000..97a1ed4 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t5_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6974641680717468}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9835, "eval_loss": 0.15209300994873046, "wall_s": 5.85512113571167, "val_acc": 0.9835, "val_loss": 0.15209300994873046, "cos_fa_negg": [0.8636005520820618, 0.9084952473640442, 0.3000030815601349, 0.9496800303459167], "loss": 0.1541658639907837, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.6851248741149902, 0.29962611198425293, 0.45948222279548645], "lesion_eval_acc": 0.7445, "lesion_eval_loss": 0.4298059539794922, "lesion_acc_drop": 0.239}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 5.720849275588989, "diagnostics_wall_s": 0.07393050193786621, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 5.720849275588989, "evaluation_wall_s": 0.05547380447387695}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s4.json b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s4.json new file mode 100644 index 0000000..fa5b007 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_fa_d4_t5_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "fa", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.02, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "linear", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_fa_d4_t5_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 1.0589196681976318}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.7115, "eval_loss": 0.5537738952636718, "wall_s": 8.392360210418701, "val_acc": 0.7115, "val_loss": 0.5537738952636718, "cos_fa_negg": [0.818917989730835, 0.7523933053016663, 0.9636358022689819, 0.9373975992202759], "loss": 0.5792245268821716, "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.5218450427055359, 0.10892205685377121, 0.2169610559940338], "lesion_eval_acc": 0.639, "lesion_eval_loss": 1.1532683715820313, "lesion_acc_drop": 0.07250000000000001}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 8.26922059059143, "diagnostics_wall_s": 0.06295275688171387, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 8.26922059059143, "evaluation_wall_s": 0.05523490905761719}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 0, "calibration_batch_loss_evaluations": 0, "calibration_example_loss_evaluations": 0, "calibration_forward_equivalent_examples": 0.0, "training_forward_equivalent_examples": 640000.0, "max_perturbation_batch_expansion": 0}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17540608, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s0.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s0.json new file mode 100644 index 0000000..9f0864c --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7108338475227356}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.699, "eval_loss": 0.6175665283203124, "wall_s": 4.590121030807495, "val_acc": 0.699, "val_loss": 0.6175665283203124, "cos_r_negg": [0.9992705583572388], "cos_innovation_negg": [0.9992705583572388], "cos_apical_negg": [0.9992705583572388], "cos_Ac_negg": [0.9992705583572388], "r_norm": [3.191342830657959], "g_norm": [0.007304305676370859], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.505754709243774, "diagnostics_wall_s": 0.0531158447265625, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.505754709243774, "evaluation_wall_s": 0.026494979858398438}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s1.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s1.json new file mode 100644 index 0000000..1cd1e16 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6921946406364441}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.6095, "eval_loss": 0.619910400390625, "wall_s": 4.528242588043213, "val_acc": 0.6095, "val_loss": 0.619910400390625, "cos_r_negg": [0.9995871782302856], "cos_innovation_negg": [0.9995871782302856], "cos_apical_negg": [0.9995871782302856], "cos_Ac_negg": [0.9995871782302856], "r_norm": [3.0221714973449707], "g_norm": [0.0069101732224226], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.443234205245972, "diagnostics_wall_s": 0.054076194763183594, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.443234205245972, "evaluation_wall_s": 0.026144027709960938}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s2.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s2.json new file mode 100644 index 0000000..d077fff --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7062802314758301}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.543, "eval_loss": 0.6199972534179687, "wall_s": 4.60357928276062, "val_acc": 0.543, "val_loss": 0.6199972534179687, "cos_r_negg": [0.9983969926834106], "cos_innovation_negg": [0.9983969926834106], "cos_apical_negg": [0.9983969926834106], "cos_Ac_negg": [0.9983969926834106], "r_norm": [2.9471335411071777], "g_norm": [0.006557501386851072], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.527036666870117, "diagnostics_wall_s": 0.04650259017944336, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.527036666870117, "evaluation_wall_s": 0.02507328987121582}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s3.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s3.json new file mode 100644 index 0000000..4a900ba --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7154970765113831}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5865, "eval_loss": 0.6083611145019531, "wall_s": 4.558659076690674, "val_acc": 0.5865, "val_loss": 0.6083611145019531, "cos_r_negg": [0.9985916614532471], "cos_innovation_negg": [0.9985916614532471], "cos_apical_negg": [0.9985916614532471], "cos_Ac_negg": [0.9985916614532471], "r_norm": [2.80849552154541], "g_norm": [0.006729436572641134], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.481391668319702, "diagnostics_wall_s": 0.04822421073913574, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.481391668319702, "evaluation_wall_s": 0.023919343948364258}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s4.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s4.json new file mode 100644 index 0000000..c3a7aa9 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t3_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7816296219825745}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.7225, "eval_loss": 0.5547794189453125, "wall_s": 4.83549427986145, "val_acc": 0.7225, "val_loss": 0.5547794189453125, "cos_r_negg": [0.99879390001297], "cos_innovation_negg": [0.99879390001297], "cos_apical_negg": [0.99879390001297], "cos_Ac_negg": [0.99879390001297], "r_norm": [4.071575164794922], "g_norm": [0.009064728394150734], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.756731986999512, "diagnostics_wall_s": 0.04867386817932129, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.756731986999512, "evaluation_wall_s": 0.02542281150817871}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s0.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s0.json new file mode 100644 index 0000000..5646550 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6991833448410034}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.658, "eval_loss": 0.6097836608886719, "wall_s": 4.55283260345459, "val_acc": 0.658, "val_loss": 0.6097836608886719, "cos_r_negg": [0.9993320107460022], "cos_innovation_negg": [0.9993320107460022], "cos_apical_negg": [0.9993320107460022], "cos_Ac_negg": [0.9993320107460022], "r_norm": [3.3141374588012695], "g_norm": [0.007456951774656773], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.47908091545105, "diagnostics_wall_s": 0.04493212699890137, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.47908091545105, "evaluation_wall_s": 0.024128437042236328}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s1.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s1.json new file mode 100644 index 0000000..3d92600 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6932581663131714}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5105, "eval_loss": 0.6141571350097657, "wall_s": 4.8446433544158936, "val_acc": 0.5105, "val_loss": 0.6141571350097657, "cos_r_negg": [0.9995845556259155], "cos_innovation_negg": [0.9995845556259155], "cos_apical_negg": [0.9995845556259155], "cos_Ac_negg": [0.9995845556259155], "r_norm": [3.1208274364471436], "g_norm": [0.007033518049865961], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.768789291381836, "diagnostics_wall_s": 0.04642200469970703, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.768789291381836, "evaluation_wall_s": 0.02473139762878418}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s2.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s2.json new file mode 100644 index 0000000..7292410 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6969773173332214}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5245, "eval_loss": 0.6149517822265625, "wall_s": 4.6266772747039795, "val_acc": 0.5245, "val_loss": 0.6149517822265625, "cos_r_negg": [0.9984625577926636], "cos_innovation_negg": [0.9984625577926636], "cos_apical_negg": [0.9984625577926636], "cos_Ac_negg": [0.9984625577926636], "r_norm": [2.9811718463897705], "g_norm": [0.006587645038962364], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.543834447860718, "diagnostics_wall_s": 0.05278754234313965, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.543834447860718, "evaluation_wall_s": 0.02538585662841797}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s3.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s3.json new file mode 100644 index 0000000..1b108b2 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.701652467250824}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5, "eval_loss": 0.6946137084960937, "wall_s": 4.672075033187866, "val_acc": 0.5, "val_loss": 0.6946137084960937, "cos_r_negg": [0.9937993288040161], "cos_innovation_negg": [0.9937993288040161], "cos_apical_negg": [0.9937993288040161], "cos_Ac_negg": [0.9937993288040161], "r_norm": [0.44873467087745667], "g_norm": [0.0007474047597497702], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.586468935012817, "diagnostics_wall_s": 0.053025245666503906, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.586468935012817, "evaluation_wall_s": 0.027146339416503906}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s4.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s4.json new file mode 100644 index 0000000..6a1ef70 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t4_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.8244266510009766}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.7015, "eval_loss": 0.5561688537597657, "wall_s": 4.797708988189697, "val_acc": 0.7015, "val_loss": 0.5561688537597657, "cos_r_negg": [0.9986674785614014], "cos_innovation_negg": [0.9986674785614014], "cos_apical_negg": [0.9986674785614014], "cos_Ac_negg": [0.9986674785614014], "r_norm": [4.166697025299072], "g_norm": [0.009286968037486076], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.710899114608765, "diagnostics_wall_s": 0.04720616340637207, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.710899114608765, "evaluation_wall_s": 0.024312257766723633}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s0.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s0.json new file mode 100644 index 0000000..6d2f082 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7017548084259033}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.6845, "eval_loss": 0.6106488037109375, "wall_s": 4.591655731201172, "val_acc": 0.6845, "val_loss": 0.6106488037109375, "cos_r_negg": [0.9992144703865051], "cos_innovation_negg": [0.9992144703865051], "cos_apical_negg": [0.9992144703865051], "cos_Ac_negg": [0.9992144703865051], "r_norm": [3.198115825653076], "g_norm": [0.007272394374012947], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.512455701828003, "diagnostics_wall_s": 0.04984760284423828, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.512455701828003, "evaluation_wall_s": 0.024669170379638672}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s1.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s1.json new file mode 100644 index 0000000..a8607f0 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6929494142532349}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.67, "eval_loss": 0.6155944519042968, "wall_s": 4.594546318054199, "val_acc": 0.67, "val_loss": 0.6155944519042968, "cos_r_negg": [0.9996352195739746], "cos_innovation_negg": [0.9996352195739746], "cos_apical_negg": [0.9996352195739746], "cos_Ac_negg": [0.9996352195739746], "r_norm": [3.036874771118164], "g_norm": [0.0068736448884010315], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.500474691390991, "diagnostics_wall_s": 0.0646817684173584, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.500474691390991, "evaluation_wall_s": 0.024770021438598633}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s2.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s2.json new file mode 100644 index 0000000..ddc2431 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6991405487060547}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.726, "eval_loss": 0.6109078063964843, "wall_s": 4.708315849304199, "val_acc": 0.726, "val_loss": 0.6109078063964843, "cos_r_negg": [0.9990293979644775], "cos_innovation_negg": [0.9990293979644775], "cos_apical_negg": [0.9990293979644775], "cos_Ac_negg": [0.9990293979644775], "r_norm": [2.918693780899048], "g_norm": [0.006496204063296318], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.631200313568115, "diagnostics_wall_s": 0.04613089561462402, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.631200313568115, "evaluation_wall_s": 0.02622079849243164}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s3.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s3.json new file mode 100644 index 0000000..ae229b8 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7045606374740601}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.7155, "eval_loss": 0.5956400756835938, "wall_s": 4.46791934967041, "val_acc": 0.7155, "val_loss": 0.5956400756835938, "cos_r_negg": [0.9982938766479492], "cos_innovation_negg": [0.9982938766479492], "cos_apical_negg": [0.9982938766479492], "cos_Ac_negg": [0.9982938766479492], "r_norm": [2.6968917846679688], "g_norm": [0.006479310803115368], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.388706922531128, "diagnostics_wall_s": 0.04989790916442871, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.388706922531128, "evaluation_wall_s": 0.024506330490112305}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s4.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s4.json new file mode 100644 index 0000000..130be86 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 1, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.0, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d1_t5_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7986906170845032}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.7455, "eval_loss": 0.5351124877929687, "wall_s": 4.498694658279419, "val_acc": 0.7455, "val_loss": 0.5351124877929687, "cos_r_negg": [0.9987180829048157], "cos_innovation_negg": [0.9987180829048157], "cos_apical_negg": [0.9987180829048157], "cos_Ac_negg": [0.9987180829048157], "r_norm": [3.9851267337799072], "g_norm": [0.008898783475160599], "traffic_norm": [0.0], "traffic_residual_norm": [0.0], "traffic_r2": [NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 4.422644138336182, "diagnostics_wall_s": 0.047634124755859375, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 4.422644138336182, "evaluation_wall_s": 0.0239260196685791}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17396736, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s0.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s0.json new file mode 100644 index 0000000..96b55f3 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6894387006759644}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9705, "eval_loss": 0.08722280883789063, "wall_s": 12.11314582824707, "val_acc": 0.9705, "val_loss": 0.08722280883789063, "cos_r_negg": [0.5163539052009583, 0.7009778618812561, 0.5394864082336426, 0.9879360198974609], "cos_innovation_negg": [0.5163539052009583, 0.7009778618812561, 0.5394864082336426, 0.9879360198974609], "cos_apical_negg": [0.5163539052009583, 0.7009778618812561, 0.5394864082336426, 0.9879360198974609], "cos_Ac_negg": [0.5163539052009583, 0.7009778618812561, 0.5394864082336426, 0.9879360198974609], "r_norm": [1.9185512065887451, 1.9110920429229736, 1.368293046951294, 0.9719609618186951], "g_norm": [0.006854870356619358, 0.005115547217428684, 0.0035284615587443113, 0.0021322788670659065], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [1.101384162902832, 1.1662038564682007, 0.8024768233299255], "lesion_eval_acc": 0.5, "lesion_eval_loss": 8.86368798828125, "lesion_acc_drop": 0.47050000000000003}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 11.998846530914307, "diagnostics_wall_s": 0.06833696365356445, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 11.998846530914307, "evaluation_wall_s": 0.04033708572387695}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s1.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s1.json new file mode 100644 index 0000000..87727b7 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.745452880859375}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5, "eval_loss": NaN, "wall_s": 12.643994331359863, "val_acc": 0.5, "val_loss": NaN, "cos_r_negg": [NaN, NaN, NaN, NaN], "cos_innovation_negg": [NaN, NaN, NaN, NaN], "cos_apical_negg": [NaN, NaN, NaN, NaN], "cos_Ac_negg": [NaN, NaN, NaN, NaN], "r_norm": [NaN, NaN, NaN, NaN], "g_norm": [NaN, NaN, NaN, NaN], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [NaN, NaN, NaN, NaN], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [NaN, NaN, NaN], "lesion_eval_acc": 0.5, "lesion_eval_loss": NaN, "lesion_acc_drop": 0.0}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.525137424468994, "diagnostics_wall_s": 0.07190370559692383, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.525137424468994, "evaluation_wall_s": 0.04202866554260254}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s2.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s2.json new file mode 100644 index 0000000..1cbe421 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6903389096260071}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.965, "eval_loss": 0.11363253784179687, "wall_s": 12.923290491104126, "val_acc": 0.965, "val_loss": 0.11363253784179687, "cos_r_negg": [0.5429419279098511, 0.49443918466567993, 0.94899582862854, 0.3704104423522949], "cos_innovation_negg": [0.5429419279098511, 0.49443918466567993, 0.94899582862854, 0.3704104423522949], "cos_apical_negg": [0.5429419279098511, 0.49443918466567993, 0.94899582862854, 0.3704104423522949], "cos_Ac_negg": [0.5429419279098511, 0.49443918466567993, 0.94899582862854, 0.3704104423522949], "r_norm": [1.4657930135726929, 2.525649309158325, 1.7411102056503296, 0.7762597799301147], "g_norm": [0.004996730014681816, 0.00448640389367938, 0.0024356236681342125, 0.001412177225574851], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.6112812757492065, 1.1300896406173706, 1.007940411567688], "lesion_eval_acc": 0.5575, "lesion_eval_loss": 6.887627197265625, "lesion_acc_drop": 0.4075}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.808303594589233, "diagnostics_wall_s": 0.06798148155212402, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.808303594589233, "evaluation_wall_s": 0.04170536994934082}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s3.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s3.json new file mode 100644 index 0000000..8be56de --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7031912803649902}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.811, "eval_loss": 0.3326073455810547, "wall_s": 12.668920993804932, "val_acc": 0.811, "val_loss": 0.3326073455810547, "cos_r_negg": [0.4227507710456848, 0.8514126539230347, 0.5685356259346008, 0.9905080795288086], "cos_innovation_negg": [0.4227507710456848, 0.8514126539230347, 0.5685356259346008, 0.9905080795288086], "cos_apical_negg": [0.4227507710456848, 0.8514126539230347, 0.5685356259346008, 0.9905080795288086], "cos_Ac_negg": [0.4227507710456848, 0.8514126539230347, 0.5685356259346008, 0.9905080795288086], "r_norm": [3.1955761909484863, 5.127671241760254, 4.1847429275512695, 2.726026773452759], "g_norm": [0.012001147493720055, 0.011348895728588104, 0.009653301909565926, 0.005646109580993652], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.5754694938659668, 0.23706328868865967, 0.9111143946647644], "lesion_eval_acc": 0.459, "lesion_eval_loss": 3.481731689453125, "lesion_acc_drop": 0.35200000000000004}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.553369760513306, "diagnostics_wall_s": 0.07114052772521973, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.553369760513306, "evaluation_wall_s": 0.0394129753112793}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s4.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s4.json new file mode 100644 index 0000000..a373440 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 3, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t3_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 3, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "fdedfe8d9534c1d64d0ac2920c7b61b8e1ff98e9b98c326d64f12f07c65dfc18", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 1.1787136793136597}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.937, "eval_loss": 0.42631959533691405, "wall_s": 12.801405668258667, "val_acc": 0.937, "val_loss": 0.42631959533691405, "cos_r_negg": [0.8638493418693542, 0.9314695000648499, 0.3915325105190277, 0.9894595146179199], "cos_innovation_negg": [0.8638493418693542, 0.9314695000648499, 0.3915325105190277, 0.9894595146179199], "cos_apical_negg": [0.8638493418693542, 0.9314695000648499, 0.3915325105190277, 0.9894595146179199], "cos_Ac_negg": [0.8638493418693542, 0.9314695000648499, 0.3915325105190277, 0.9894595146179199], "r_norm": [2.821139335632324, 2.592513084411621, 2.4672138690948486, 2.3995983600616455], "g_norm": [0.007452340796589851, 0.006937646307051182, 0.006548257078975439, 0.0057966504245996475], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.588287889957428, 0.21865318715572357, 0.4341731667518616], "lesion_eval_acc": 0.5, "lesion_eval_loss": 0.9159993896484375, "lesion_acc_drop": 0.43700000000000006}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.686410665512085, "diagnostics_wall_s": 0.06761837005615234, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.686410665512085, "evaluation_wall_s": 0.04289579391479492}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s0.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s0.json new file mode 100644 index 0000000..262215e --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6999248266220093}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.974, "eval_loss": 0.10272121810913086, "wall_s": 12.980058908462524, "val_acc": 0.974, "val_loss": 0.10272121810913086, "cos_r_negg": [0.5782746076583862, 0.8191092014312744, 0.8033065795898438, 0.9992046356201172], "cos_innovation_negg": [0.5782746076583862, 0.8191092014312744, 0.8033065795898438, 0.9992046356201172], "cos_apical_negg": [0.5782746076583862, 0.8191092014312744, 0.8033065795898438, 0.9992046356201172], "cos_Ac_negg": [0.5782746076583862, 0.8191092014312744, 0.8033065795898438, 0.9992046356201172], "r_norm": [1.3080933094024658, 1.4547786712646484, 1.1260263919830322, 0.9082434773445129], "g_norm": [0.00473866518586874, 0.0039556436240673065, 0.002893520053476095, 0.002165459096431732], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.9404135346412659, 1.0159721374511719, 0.5982341170310974], "lesion_eval_acc": 0.5025, "lesion_eval_loss": 9.43493603515625, "lesion_acc_drop": 0.47150000000000003}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.861948490142822, "diagnostics_wall_s": 0.06994438171386719, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.861948490142822, "evaluation_wall_s": 0.04251408576965332}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s1.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s1.json new file mode 100644 index 0000000..c4c9cbf --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.7811087965965271}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5, "eval_loss": 3.4855672607421875, "wall_s": 12.240322589874268, "val_acc": 0.5, "val_loss": 3.4855672607421875, "cos_r_negg": [0.945831298828125, -0.29371488094329834, 0.2264617681503296, 0.10369207710027695], "cos_innovation_negg": [0.945831298828125, -0.29371488094329834, 0.2264617681503296, 0.10369207710027695], "cos_apical_negg": [0.945831298828125, -0.29371488094329834, 0.2264617681503296, 0.10369207710027695], "cos_Ac_negg": [0.945831298828125, -0.29371488094329834, 0.2264617681503296, 0.10369207710027695], "r_norm": [1.8412928581237793, 1.5582826137542725, 1.7451380491256714, 1.593011498451233], "g_norm": [0.005650863517075777, 0.004104212392121553, 0.004101465921849012, 0.004180764779448509], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.8425371050834656, 0.6801510453224182, 2.2257089614868164], "lesion_eval_acc": 0.5, "lesion_eval_loss": 3.05464208984375, "lesion_acc_drop": 0.0}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.121156215667725, "diagnostics_wall_s": 0.07338261604309082, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.121156215667725, "evaluation_wall_s": 0.04102063179016113}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s2.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s2.json new file mode 100644 index 0000000..f465910 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6940909624099731}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5, "eval_loss": 1.4991718794914362e+31, "wall_s": 12.378154277801514, "val_acc": 0.5, "val_loss": 1.4991718794914362e+31, "cos_r_negg": [1.0, NaN, NaN, NaN], "cos_innovation_negg": [1.0, NaN, NaN, NaN], "cos_apical_negg": [1.0, NaN, NaN, NaN], "cos_Ac_negg": [1.0, NaN, NaN, NaN], "r_norm": [3.804513384058061e+16, 3.3507086428012544e+16, 3.366338243539763e+16, 2.9594472153088e+16], "g_norm": [9.270828642336768e+16, 9.270828642336768e+16, 9.270828642336768e+16, 9.270828642336768e+16], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.0, 0.0, 0.0], "lesion_eval_acc": 0.5, "lesion_eval_loss": 1.4991718794914362e+31, "lesion_acc_drop": 0.0}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.259091138839722, "diagnostics_wall_s": 0.07349634170532227, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.259091138839722, "evaluation_wall_s": 0.04076361656188965}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s3.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s3.json new file mode 100644 index 0000000..b0bbcf7 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6956391334533691}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.993, "eval_loss": 0.04431462097167969, "wall_s": 12.185118198394775, "val_acc": 0.993, "val_loss": 0.04431462097167969, "cos_r_negg": [0.6834122538566589, 0.7956693172454834, 0.9033420085906982, 0.9615830183029175], "cos_innovation_negg": [0.6834122538566589, 0.7956693172454834, 0.9033420085906982, 0.9615830183029175], "cos_apical_negg": [0.6834122538566589, 0.7956693172454834, 0.9033420085906982, 0.9615830183029175], "cos_Ac_negg": [0.6834122538566589, 0.7956693172454834, 0.9033420085906982, 0.9615830183029175], "r_norm": [1.14504075050354, 0.5734323263168335, 0.5399302840232849, 0.2471168339252472], "g_norm": [0.002859368221834302, 0.0013482656795531511, 0.0010114673059433699, 0.0005220217863097787], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [1.0089523792266846, 0.7032589316368103, 0.6244146227836609], "lesion_eval_acc": 0.5, "lesion_eval_loss": 9.0192822265625, "lesion_acc_drop": 0.493}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.067923784255981, "diagnostics_wall_s": 0.0681917667388916, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.067923784255981, "evaluation_wall_s": 0.043576955795288086}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s4.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s4.json new file mode 100644 index 0000000..5bb7ecf --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 4, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t4_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 4, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "9e82c4f08ab216367fb95592e461929e3aecf2d9041f5e7756f77d140378bd3a", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 1.0611846446990967}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.958, "eval_loss": 0.13668526458740235, "wall_s": 12.70658564567566, "val_acc": 0.958, "val_loss": 0.13668526458740235, "cos_r_negg": [0.9715449810028076, 0.9857341051101685, 0.9910546541213989, 0.9814310669898987], "cos_innovation_negg": [0.9715449810028076, 0.9857341051101685, 0.9910546541213989, 0.9814310669898987], "cos_apical_negg": [0.9715449810028076, 0.9857341051101685, 0.9910546541213989, 0.9814310669898987], "cos_Ac_negg": [0.9715449810028076, 0.9857341051101685, 0.9910546541213989, 0.9814310669898987], "r_norm": [1.3124041557312012, 0.9584861993789673, 0.7369867563247681, 0.5346509218215942], "g_norm": [0.011630582623183727, 0.0079872477799654, 0.004587069619446993, 0.0020537609234452248], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.5864320993423462, 0.13167251646518707, 0.3070070147514343], "lesion_eval_acc": 0.593, "lesion_eval_loss": 0.7722555541992188, "lesion_acc_drop": 0.365}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.5887291431427, "diagnostics_wall_s": 0.06946897506713867, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.5887291431427, "evaluation_wall_s": 0.041778564453125}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s0.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s0.json new file mode 100644 index 0000000..ac63c49 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s0", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6962454319000244}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9435, "eval_loss": 0.11786185455322265, "wall_s": 12.59320878982544, "val_acc": 0.9435, "val_loss": 0.11786185455322265, "cos_r_negg": [0.5609173774719238, 0.775170624256134, 0.8045323491096497, 0.9909271001815796], "cos_innovation_negg": [0.5609173774719238, 0.775170624256134, 0.8045323491096497, 0.9909271001815796], "cos_apical_negg": [0.5609173774719238, 0.775170624256134, 0.8045323491096497, 0.9909271001815796], "cos_Ac_negg": [0.5609173774719238, 0.775170624256134, 0.8045323491096497, 0.9909271001815796], "r_norm": [1.6331727504730225, 1.6468857526779175, 1.117485761642456, 0.9248777031898499], "g_norm": [0.0066820178180933, 0.005063867196440697, 0.0030614794231951237, 0.0022790860384702682], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.9719088077545166, 1.108108639717102, 0.8111305832862854], "lesion_eval_acc": 0.5, "lesion_eval_loss": 10.97798876953125, "lesion_acc_drop": 0.4435}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.467562675476074, "diagnostics_wall_s": 0.08104610443115234, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.467562675476074, "evaluation_wall_s": 0.03981304168701172}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s1.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s1.json new file mode 100644 index 0000000..db86d1c --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s1.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 1, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s1", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.761730432510376}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.5, "eval_loss": 2.603609375, "wall_s": 12.879198789596558, "val_acc": 0.5, "val_loss": 2.603609375, "cos_r_negg": [-0.3716849088668823, 0.7735829949378967, -0.7593997120857239, -0.9989346265792847], "cos_innovation_negg": [-0.3716849088668823, 0.7735829949378967, -0.7593997120857239, -0.9989346265792847], "cos_apical_negg": [-0.3716849088668823, 0.7735829949378967, -0.7593997120857239, -0.9989346265792847], "cos_Ac_negg": [-0.3716849088668823, 0.7735829949378967, -0.7593997120857239, -0.9989346265792847], "r_norm": [0.8271975517272949, 1.0369493961334229, 0.826537013053894, 0.7035311460494995], "g_norm": [0.0015109528321772814, 0.0013157965149730444, 0.0011528771137818694, 0.0009299375815317035], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.7565590739250183, 1.3959684371948242, 1.392912745475769], "lesion_eval_acc": 0.5, "lesion_eval_loss": 2.08743896484375, "lesion_acc_drop": 0.0}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.74432373046875, "diagnostics_wall_s": 0.07416629791259766, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.74432373046875, "evaluation_wall_s": 0.044660091400146484}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s2.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s2.json new file mode 100644 index 0000000..fe147ed --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s2.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 2, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s2", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.6929397583007812}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.4295, "eval_loss": 5.69566748046875, "wall_s": 12.755396604537964, "val_acc": 0.4295, "val_loss": 5.69566748046875, "cos_r_negg": [0.6138091087341309, NaN, -0.4189910888671875, -0.5389285683631897], "cos_innovation_negg": [0.6138091087341309, NaN, -0.4189910888671875, -0.5389285683631897], "cos_apical_negg": [0.6138091087341309, NaN, -0.4189910888671875, -0.5389285683631897], "cos_Ac_negg": [0.6138091087341309, NaN, -0.4189910888671875, -0.5389285683631897], "r_norm": [7.728099346160889, 7.379948616027832, 7.304427146911621, 5.896245002746582], "g_norm": [0.011582979001104832, 0.011582979001104832, 0.009197389706969261, 0.014116710051894188], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.0, 2.423720121383667, 6.0582685470581055], "lesion_eval_acc": 0.5, "lesion_eval_loss": 41.220462890625, "lesion_acc_drop": -0.07050000000000001}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.644466161727905, "diagnostics_wall_s": 0.06672167778015137, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.644466161727905, "evaluation_wall_s": 0.039583683013916016}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s3.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s3.json new file mode 100644 index 0000000..96c5e24 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s3.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 3, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s3", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 0.697464108467102}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.9345, "eval_loss": 0.12317059707641602, "wall_s": 12.149521827697754, "val_acc": 0.9345, "val_loss": 0.12317059707641602, "cos_r_negg": [0.42019540071487427, 0.4223843514919281, 0.9887145757675171, 0.979535698890686], "cos_innovation_negg": [0.42019540071487427, 0.4223843514919281, 0.9887145757675171, 0.979535698890686], "cos_apical_negg": [0.42019540071487427, 0.4223843514919281, 0.9887145757675171, 0.979535698890686], "cos_Ac_negg": [0.42019540071487427, 0.4223843514919281, 0.9887145757675171, 0.979535698890686], "r_norm": [0.6567388772964478, 0.5049489736557007, 0.9152029156684875, 0.7174856662750244], "g_norm": [0.0041768853552639484, 0.00281762657687068, 0.002475907327607274, 0.0015150278341025114], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.6851263642311096, 1.2819403409957886, 0.3362151086330414], "lesion_eval_acc": 0.5, "lesion_eval_loss": 5.745776611328125, "lesion_acc_drop": 0.4345}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.025551795959473, "diagnostics_wall_s": 0.0776827335357666, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.025551795959473, "evaluation_wall_s": 0.041785478591918945}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file diff --git a/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s4.json b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s4.json new file mode 100644 index 0000000..77f34c6 --- /dev/null +++ b/results/c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s4.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "tentmap", "depth": 4, "width": 8, "act": "relu", "residual": 1, "residual_lesion_fraction": 0.3333333333333333, "epochs": 80, "batch_size": 256, "train_examples": 0, "val_examples": 2000, "split_seed": 2027, "eval_split": "validation", "task_seed": 5, "task_train_examples": 10000, "task_test_examples": 5000, "task_levels": 2, "task_n_in": 1, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.03, "eta_A": 0.01, "eta_P": 0.002, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "vectorizer_mode": "context_gated", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 4, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c2_context_val_v1_tent_l2_w8_sdil_d4_t5_s4", "n_in": 1, "n_out": 2}, "split": {"dataset": "tentmap", "task_seed": 5, "train_examples": 8000, "test_examples": 5000, "evaluation_split": "validation", "synthetic_generator": true, "split_seed": 2027, "validation_examples": 2000, "validation_index_sha256": "2335c66a37fdb438e9b862f9f7047842ab8a843875e1727368e9499ca6b633e8", "validation_class_counts": {"0": 1000, "1": 1000}, "split_from_training_only": true}, "provenance": {"git_commit": "b18227f5534284d33ec4f7b0943c90573473c6e7", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 1.0589196681976318}, {"epoch_end": 0, "step": 32}, {"epoch_end": 1, "step": 64}, {"epoch_end": 2, "step": 96}, {"epoch_end": 3, "step": 128}, {"epoch_end": 4, "step": 160}, {"epoch_end": 5, "step": 192}, {"epoch_end": 6, "step": 224}, {"epoch_end": 7, "step": 256}, {"epoch_end": 8, "step": 288}, {"epoch_end": 9, "step": 320}, {"epoch_end": 10, "step": 352}, {"epoch_end": 11, "step": 384}, {"epoch_end": 12, "step": 416}, {"epoch_end": 13, "step": 448}, {"epoch_end": 14, "step": 480}, {"epoch_end": 15, "step": 512}, {"epoch_end": 16, "step": 544}, {"epoch_end": 17, "step": 576}, {"epoch_end": 18, "step": 608}, {"epoch_end": 19, "step": 640}, {"epoch_end": 20, "step": 672}, {"epoch_end": 21, "step": 704}, {"epoch_end": 22, "step": 736}, {"epoch_end": 23, "step": 768}, {"epoch_end": 24, "step": 800}, {"epoch_end": 25, "step": 832}, {"epoch_end": 26, "step": 864}, {"epoch_end": 27, "step": 896}, {"epoch_end": 28, "step": 928}, {"epoch_end": 29, "step": 960}, {"epoch_end": 30, "step": 992}, {"epoch_end": 31, "step": 1024}, {"epoch_end": 32, "step": 1056}, {"epoch_end": 33, "step": 1088}, {"epoch_end": 34, "step": 1120}, {"epoch_end": 35, "step": 1152}, {"epoch_end": 36, "step": 1184}, {"epoch_end": 37, "step": 1216}, {"epoch_end": 38, "step": 1248}, {"epoch_end": 39, "step": 1280}, {"epoch_end": 40, "step": 1312}, {"epoch_end": 41, "step": 1344}, {"epoch_end": 42, "step": 1376}, {"epoch_end": 43, "step": 1408}, {"epoch_end": 44, "step": 1440}, {"epoch_end": 45, "step": 1472}, {"epoch_end": 46, "step": 1504}, {"epoch_end": 47, "step": 1536}, {"epoch_end": 48, "step": 1568}, {"epoch_end": 49, "step": 1600}, {"epoch_end": 50, "step": 1632}, {"epoch_end": 51, "step": 1664}, {"epoch_end": 52, "step": 1696}, {"epoch_end": 53, "step": 1728}, {"epoch_end": 54, "step": 1760}, {"epoch_end": 55, "step": 1792}, {"epoch_end": 56, "step": 1824}, {"epoch_end": 57, "step": 1856}, {"epoch_end": 58, "step": 1888}, {"epoch_end": 59, "step": 1920}, {"epoch_end": 60, "step": 1952}, {"epoch_end": 61, "step": 1984}, {"epoch_end": 62, "step": 2016}, {"epoch_end": 63, "step": 2048}, {"epoch_end": 64, "step": 2080}, {"epoch_end": 65, "step": 2112}, {"epoch_end": 66, "step": 2144}, {"epoch_end": 67, "step": 2176}, {"epoch_end": 68, "step": 2208}, {"epoch_end": 69, "step": 2240}, {"epoch_end": 70, "step": 2272}, {"epoch_end": 71, "step": 2304}, {"epoch_end": 72, "step": 2336}, {"epoch_end": 73, "step": 2368}, {"epoch_end": 74, "step": 2400}, {"epoch_end": 75, "step": 2432}, {"epoch_end": 76, "step": 2464}, {"epoch_end": 77, "step": 2496}, {"epoch_end": 78, "step": 2528}, {"epoch_end": 79, "step": 2560}], "final": {"eval_split": "validation", "eval_acc": 0.952, "eval_loss": 0.36534596252441404, "wall_s": 13.017056226730347, "val_acc": 0.952, "val_loss": 0.36534596252441404, "cos_r_negg": [0.7939774394035339, 0.9271726012229919, 0.5010230541229248, 0.9880345463752747], "cos_innovation_negg": [0.7939774394035339, 0.9271726012229919, 0.5010230541229248, 0.9880345463752747], "cos_apical_negg": [0.7939774394035339, 0.9271726012229919, 0.5010230541229248, 0.9880345463752747], "cos_Ac_negg": [0.7939774394035339, 0.9271726012229919, 0.5010230541229248, 0.9880345463752747], "r_norm": [2.671567916870117, 2.4010508060455322, 2.2658321857452393, 2.2099950313568115], "g_norm": [0.007250039838254452, 0.006702700164169073, 0.006319956388324499, 0.005611311178654432], "traffic_norm": [0.0, 0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN, NaN], "residual_lesion": {"interior_layers": [1, 2, 3], "lesioned_layers": [3], "branch_to_skip_rms": [0.6111488938331604, 0.17128272354602814, 0.36644285917282104], "lesion_eval_acc": 0.6835, "lesion_eval_loss": 0.8940163269042969, "lesion_acc_drop": 0.26849999999999996}}, "lesion_protocol": {"fraction_of_interior_blocks": 0.3333333333333333, "selection": "final_contiguous_interior_blocks", "probe_source": "training_prefix", "probe_examples": 512, "evaluation_split": "validation"}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 12.900709629058838, "diagnostics_wall_s": 0.07080912590026855, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 12.900709629058838, "evaluation_wall_s": 0.040711402893066406}, "cost": {"train_steps": 2560, "ordinary_training_forward_examples": 640000, "predictor_warmup_forward_examples": 0, "perturbation_events": 640, "calibration_batch_loss_evaluations": 1280, "calibration_example_loss_evaluations": 327680, "calibration_forward_equivalent_examples": 491520.0, "training_forward_equivalent_examples": 1131520.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 17590784, "peak_memory_reserved_bytes": 23068672}} \ No newline at end of file -- cgit v1.2.3