diff options
Diffstat (limited to 'results')
27 files changed, 27 insertions, 0 deletions
diff --git a/results/c1_fmnist_val_v1_none_rho0_t1234_matched_d3_s0.json b/results/c1_fmnist_val_v1_none_rho0_t1234_matched_d3_s0.json new file mode 100644 index 0000000..8e673e9 --- /dev/null +++ b/results/c1_fmnist_val_v1_none_rho0_t1234_matched_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "match_innovation_norm", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_none_rho0_t1234_matched_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.502072334289551}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.88, "eval_loss": 0.41392838134765625, "wall_s": 22.086261987686157, "val_acc": 0.88, "val_loss": 0.41392838134765625, "cos_r_negg": [0.8215733766555786, 0.5133273601531982, 0.4824512004852295], "cos_innovation_negg": [0.8215733766555786, 0.5133273601531982, 0.4824512004852295], "cos_apical_negg": [0.8215733766555786, 0.5133273601531982, 0.4824512004852295], "cos_Ac_negg": [0.8215733766555786, 0.5133273601531982, 0.4824512004852295], "r_norm": [0.49292951822280884, 0.4109225571155548, 0.41594964265823364], "g_norm": [0.0011965048033744097, 0.0011563380248844624, 0.0011454486520960927], "traffic_norm": [0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 22.016796827316284, "diagnostics_wall_s": 0.04057931900024414, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 22.016796827316284, "evaluation_wall_s": 0.027411460876464844}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 0, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1444392.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219243008, "peak_memory_reserved_bytes": 236978176}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_none_rho0_t1234_raw_d3_s0.json b/results/c1_fmnist_val_v1_none_rho0_t1234_raw_d3_s0.json new file mode 100644 index 0000000..5c9f006 --- /dev/null +++ b/results/c1_fmnist_val_v1_none_rho0_t1234_raw_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_none_rho0_t1234_raw_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.502072334289551}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.88, "eval_loss": 0.41392838134765625, "wall_s": 20.670273065567017, "val_acc": 0.88, "val_loss": 0.41392838134765625, "cos_r_negg": [0.8215733766555786, 0.5133273601531982, 0.4824512004852295], "cos_innovation_negg": [0.8215733766555786, 0.5133273601531982, 0.4824512004852295], "cos_apical_negg": [0.8215733766555786, 0.5133273601531982, 0.4824512004852295], "cos_Ac_negg": [0.8215733766555786, 0.5133273601531982, 0.4824512004852295], "r_norm": [0.49292951822280884, 0.4109225571155548, 0.41594964265823364], "g_norm": [0.0011965048033744097, 0.0011563380248844624, 0.0011454486520960927], "traffic_norm": [0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 20.555959939956665, "diagnostics_wall_s": 0.07414674758911133, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 20.555959939956665, "evaluation_wall_s": 0.031392812728881836}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 0, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1444392.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 218714624, "peak_memory_reserved_bytes": 236978176}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_none_rho0_t1234_residual_d3_s0.json b/results/c1_fmnist_val_v1_none_rho0_t1234_residual_d3_s0.json new file mode 100644 index 0000000..212cfd4 --- /dev/null +++ b/results/c1_fmnist_val_v1_none_rho0_t1234_residual_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.0, "traffic_seed": 1234, "traffic_mode": "none", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_none_rho0_t1234_residual_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.502072334289551}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.88, "eval_loss": 0.41392838134765625, "wall_s": 20.260318517684937, "val_acc": 0.88, "val_loss": 0.41392838134765625, "cos_r_negg": [0.8215733766555786, 0.5133273601531982, 0.4824512004852295], "cos_innovation_negg": [0.8215733766555786, 0.5133273601531982, 0.4824512004852295], "cos_apical_negg": [0.8215733766555786, 0.5133273601531982, 0.4824512004852295], "cos_Ac_negg": [0.8215733766555786, 0.5133273601531982, 0.4824512004852295], "r_norm": [0.49292951822280884, 0.4109225571155548, 0.41594964265823364], "g_norm": [0.0011965048033744097, 0.0011563380248844624, 0.0011454486520960927], "traffic_norm": [0.0, 0.0, 0.0], "traffic_residual_norm": [0.0, 0.0, 0.0], "traffic_r2": [NaN, NaN, NaN]}, "timing": {"warmup_wall_s": 0.0, "training_loop_wall_s": 20.172735691070557, "diagnostics_wall_s": 0.06032729148864746, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 20.172735691070557, "evaluation_wall_s": 0.025804519653320312}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 0, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1444392.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 218714624, "peak_memory_reserved_bytes": 236978176}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_soma_rho0p5_t1234_matched_d3_s0.json b/results/c1_fmnist_val_v1_soma_rho0p5_t1234_matched_d3_s0.json new file mode 100644 index 0000000..fa03a69 --- /dev/null +++ b/results/c1_fmnist_val_v1_soma_rho0p5_t1234_matched_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "match_innovation_norm", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 1234, "traffic_mode": "soma", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_soma_rho0p5_t1234_matched_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.7644, "eval_loss": 1.175450537109375, "wall_s": 24.46480131149292, "val_acc": 0.7644, "val_loss": 1.175450537109375, "cos_r_negg": [0.19782330095767975, 0.28743764758110046, 0.21499493718147278], "cos_innovation_negg": [0.6815739870071411, 0.5965095162391663, 0.5017985105514526], "cos_apical_negg": [0.19782330095767975, 0.28743764758110046, 0.21499493718147278], "cos_Ac_negg": [0.8455286026000977, 0.7691764831542969, 0.68768310546875], "r_norm": [2.0237042903900146, 2.130772590637207, 2.319279432296753], "g_norm": [0.003588755615055561, 0.00347511051222682, 0.00345813250169158], "traffic_norm": [8.328964233398438, 13.017555236816406, 17.316818237304688], "traffic_residual_norm": [3.679889687191462e-06, 5.623085598926991e-06, 7.322518285945989e-06], "traffic_r2": [1.0, 1.0, 1.0]}, "timing": {"warmup_wall_s": 0.37581944465637207, "training_loop_wall_s": 23.98755931854248, "diagnostics_wall_s": 0.062494754791259766, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 23.98755931854248, "evaluation_wall_s": 0.03724169731140137}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219645440, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_soma_rho0p5_t1234_raw_d3_s0.json b/results/c1_fmnist_val_v1_soma_rho0p5_t1234_raw_d3_s0.json new file mode 100644 index 0000000..8b613e2 --- /dev/null +++ b/results/c1_fmnist_val_v1_soma_rho0p5_t1234_raw_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 1234, "traffic_mode": "soma", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_soma_rho0p5_t1234_raw_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.8236, "eval_loss": 0.7951713134765624, "wall_s": 22.069374561309814, "val_acc": 0.8236, "val_loss": 0.7951713134765624, "cos_r_negg": [0.1694754958152771, 0.160455584526062, 0.0821957141160965], "cos_innovation_negg": [0.6427469253540039, 0.5486112833023071, 0.4415867328643799], "cos_apical_negg": [0.1694754958152771, 0.160455584526062, 0.0821957141160965], "cos_Ac_negg": [0.7315320372581482, 0.5851202011108398, 0.6088696718215942], "r_norm": [8.578811645507812, 13.139307022094727, 17.31884765625], "g_norm": [0.00218264851719141, 0.0021238757763057947, 0.0021174666471779346], "traffic_norm": [8.347188949584961, 13.015305519104004, 17.244670867919922], "traffic_residual_norm": [4.354336851974949e-06, 6.635013505729148e-06, 7.993516192073002e-06], "traffic_r2": [1.0, 1.0, 1.0]}, "timing": {"warmup_wall_s": 0.4166140556335449, "training_loop_wall_s": 21.55690097808838, "diagnostics_wall_s": 0.0626077651977539, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 21.55690097808838, "evaluation_wall_s": 0.031691789627075195}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219117056, "peak_memory_reserved_bytes": 236978176}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_soma_rho0p5_t1234_residual_d3_s0.json b/results/c1_fmnist_val_v1_soma_rho0p5_t1234_residual_d3_s0.json new file mode 100644 index 0000000..f1c497c --- /dev/null +++ b/results/c1_fmnist_val_v1_soma_rho0p5_t1234_residual_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 1234, "traffic_mode": "soma", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_soma_rho0p5_t1234_residual_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.875, "eval_loss": 0.41233968505859375, "wall_s": 23.534036874771118, "val_acc": 0.875, "val_loss": 0.41233968505859375, "cos_r_negg": [0.733879804611206, 0.49027836322784424, 0.4058890640735626], "cos_innovation_negg": [0.733879804611206, 0.49027836322784424, 0.4058890640735626], "cos_apical_negg": [0.08222538232803345, 0.0737428218126297, 0.08905474096536636], "cos_Ac_negg": [0.7992807626724243, 0.5241587162017822, 0.4287489056587219], "r_norm": [0.4682266116142273, 0.3962087035179138, 0.39043310284614563], "g_norm": [0.0011483386624604464, 0.0010986719280481339, 0.0010787949431687593], "traffic_norm": [8.0953950881958, 9.352516174316406, 9.99079704284668], "traffic_residual_norm": [5.340886673366185e-06, 6.098325229686452e-06, 6.752318768121768e-06], "traffic_r2": [1.0, 1.0, 1.0]}, "timing": {"warmup_wall_s": 0.38501811027526855, "training_loop_wall_s": 23.04648494720459, "diagnostics_wall_s": 0.062426090240478516, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 23.04648494720459, "evaluation_wall_s": 0.038335561752319336}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219117056, "peak_memory_reserved_bytes": 236978176}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_soma_rho0p5_t1234_taskfit_d3_s0.json b/results/c1_fmnist_val_v1_soma_rho0p5_t1234_taskfit_d3_s0.json new file mode 100644 index 0000000..821c952 --- /dev/null +++ b/results/c1_fmnist_val_v1_soma_rho0p5_t1234_taskfit_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 0, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 1234, "traffic_mode": "soma", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_soma_rho0p5_t1234_taskfit_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.8672, "eval_loss": 0.40632547607421876, "wall_s": 20.212072134017944, "val_acc": 0.8672, "val_loss": 0.40632547607421876, "cos_r_negg": [0.38901251554489136, 0.29586061835289, 0.2593551278114319], "cos_innovation_negg": [0.38901251554489136, 0.29586061835289, 0.2593551278114319], "cos_apical_negg": [0.10725632309913635, 0.07197677344083786, 0.11252033710479736], "cos_Ac_negg": [0.8360591530799866, 0.5488649010658264, 0.49077731370925903], "r_norm": [1.039931297302246, 0.5107394456863403, 0.470855176448822], "g_norm": [0.0012663480592891574, 0.0012179482728242874, 0.0012064980110153556], "traffic_norm": [8.162910461425781, 9.092531204223633, 9.999863624572754], "traffic_residual_norm": [0.6600988507270813, 0.10312390327453613, 0.04290067404508591], "traffic_r2": [0.9934618473052979, 0.999871551990509, 0.9999815821647644]}, "timing": {"warmup_wall_s": 0.4180910587310791, "training_loop_wall_s": 19.68894863128662, "diagnostics_wall_s": 0.06882691383361816, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 19.68894863128662, "evaluation_wall_s": 0.034587860107421875}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219117056, "peak_memory_reserved_bytes": 236978176}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_soma_rho0p5_t5678_matched_d3_s0.json b/results/c1_fmnist_val_v1_soma_rho0p5_t5678_matched_d3_s0.json new file mode 100644 index 0000000..a0b64e2 --- /dev/null +++ b/results/c1_fmnist_val_v1_soma_rho0p5_t5678_matched_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "match_innovation_norm", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 5678, "traffic_mode": "soma", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_soma_rho0p5_t5678_matched_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.816, "eval_loss": 0.9383363037109375, "wall_s": 24.36557936668396, "val_acc": 0.816, "val_loss": 0.9383363037109375, "cos_r_negg": [0.1630924642086029, 0.3091428875923157, 0.24865606427192688], "cos_innovation_negg": [0.568183958530426, 0.5672831535339355, 0.5326230525970459], "cos_apical_negg": [0.1630924642086029, 0.3091428875923157, 0.24865607917308807], "cos_Ac_negg": [0.764899492263794, 0.7042793035507202, 0.682000994682312], "r_norm": [1.5494089126586914, 1.6613707542419434, 1.7672209739685059], "g_norm": [0.002741788513958454, 0.002664279192686081, 0.0026507314760237932], "traffic_norm": [8.340415954589844, 13.245558738708496, 17.478004455566406], "traffic_residual_norm": [3.861668119498063e-06, 6.174968348204857e-06, 7.659742550458759e-06], "traffic_r2": [1.0, 1.0, 1.0]}, "timing": {"warmup_wall_s": 0.420534610748291, "training_loop_wall_s": 23.754450798034668, "diagnostics_wall_s": 0.11057090759277344, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 23.754450798034668, "evaluation_wall_s": 0.07847070693969727}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219645440, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_soma_rho0p5_t5678_raw_d3_s0.json b/results/c1_fmnist_val_v1_soma_rho0p5_t5678_raw_d3_s0.json new file mode 100644 index 0000000..531fe64 --- /dev/null +++ b/results/c1_fmnist_val_v1_soma_rho0p5_t5678_raw_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 5678, "traffic_mode": "soma", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_soma_rho0p5_t5678_raw_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.8014, "eval_loss": 0.9824896118164063, "wall_s": 22.23854899406433, "val_acc": 0.8014, "val_loss": 0.9824896118164063, "cos_r_negg": [0.14438359439373016, 0.15992891788482666, 0.10801808536052704], "cos_innovation_negg": [0.5785732269287109, 0.5610681772232056, 0.42787614464759827], "cos_apical_negg": [0.14438359439373016, 0.15992891788482666, 0.10801808536052704], "cos_Ac_negg": [0.7364729642868042, 0.6148331165313721, 0.5364131927490234], "r_norm": [8.79173469543457, 13.670132637023926, 17.824848175048828], "g_norm": [0.0029232909437268972, 0.002848075469955802, 0.0028355198446661234], "traffic_norm": [8.358747482299805, 13.386871337890625, 17.567333221435547], "traffic_residual_norm": [4.313842509873211e-06, 6.924832632648759e-06, 8.586617695982568e-06], "traffic_r2": [1.0, 1.0, 1.0]}, "timing": {"warmup_wall_s": 0.4204719066619873, "training_loop_wall_s": 21.71327257156372, "diagnostics_wall_s": 0.07411384582519531, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 21.71327257156372, "evaluation_wall_s": 0.02914142608642578}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219117056, "peak_memory_reserved_bytes": 236978176}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_soma_rho0p5_t5678_residual_d3_s0.json b/results/c1_fmnist_val_v1_soma_rho0p5_t5678_residual_d3_s0.json new file mode 100644 index 0000000..c0dc16e --- /dev/null +++ b/results/c1_fmnist_val_v1_soma_rho0p5_t5678_residual_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 5678, "traffic_mode": "soma", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_soma_rho0p5_t5678_residual_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.8882, "eval_loss": 0.368654736328125, "wall_s": 21.269236087799072, "val_acc": 0.8882, "val_loss": 0.368654736328125, "cos_r_negg": [0.7299882769584656, 0.45444923639297485, 0.4123084545135498], "cos_innovation_negg": [0.7299882769584656, 0.45444923639297485, 0.4123084545135498], "cos_apical_negg": [0.08912729471921921, 0.061272695660591125, 0.03946426138281822], "cos_Ac_negg": [0.8094373941421509, 0.49019280076026917, 0.461851567029953], "r_norm": [0.40269869565963745, 0.3371710777282715, 0.3389732837677002], "g_norm": [0.0009679067879915237, 0.0009275713237002492, 0.0009151541162282228], "traffic_norm": [8.114797592163086, 9.657479286193848, 10.712383270263672], "traffic_residual_norm": [5.180900188861415e-06, 6.155710252642166e-06, 6.966988621570636e-06], "traffic_r2": [1.0, 1.0, 1.0]}, "timing": {"warmup_wall_s": 0.49432373046875, "training_loop_wall_s": 20.670420169830322, "diagnostics_wall_s": 0.07196664810180664, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 20.670420169830322, "evaluation_wall_s": 0.030913829803466797}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219117056, "peak_memory_reserved_bytes": 236978176}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_soma_rho0p5_t5678_taskfit_d3_s0.json b/results/c1_fmnist_val_v1_soma_rho0p5_t5678_taskfit_d3_s0.json new file mode 100644 index 0000000..16f34c6 --- /dev/null +++ b/results/c1_fmnist_val_v1_soma_rho0p5_t5678_taskfit_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 0, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 5678, "traffic_mode": "soma", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_soma_rho0p5_t5678_taskfit_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.8866, "eval_loss": 0.3651018249511719, "wall_s": 19.316948413848877, "val_acc": 0.8866, "val_loss": 0.3651018249511719, "cos_r_negg": [0.4301748275756836, 0.27763253450393677, 0.2891577482223511], "cos_innovation_negg": [0.4301748275756836, 0.27763253450393677, 0.2891577482223511], "cos_apical_negg": [0.09059552848339081, 0.07603657245635986, 0.08980832993984222], "cos_Ac_negg": [0.8295586109161377, 0.5122560858726501, 0.4109269678592682], "r_norm": [0.993767499923706, 0.6310384273529053, 0.5036736130714417], "g_norm": [0.0011061803670600057, 0.0010623302077874541, 0.001050581457093358], "traffic_norm": [8.148926734924316, 9.374341011047363, 10.366971969604492], "traffic_residual_norm": [0.6711105704307556, 0.31159621477127075, 0.16358256340026855], "traffic_r2": [0.9932188987731934, 0.9988961219787598, 0.9997513294219971]}, "timing": {"warmup_wall_s": 0.3586142063140869, "training_loop_wall_s": 18.867039680480957, "diagnostics_wall_s": 0.06217074394226074, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 18.867039680480957, "evaluation_wall_s": 0.02752089500427246}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219117056, "peak_memory_reserved_bytes": 236978176}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p2_t1234_matched_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p2_t1234_matched_d3_s0.json new file mode 100644 index 0000000..ecbee51 --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p2_t1234_matched_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "match_innovation_norm", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.2, "traffic_seed": 1234, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p2_t1234_matched_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.7966, "eval_loss": 1.1398289306640625, "wall_s": 26.187846422195435, "val_acc": 0.7966, "val_loss": 1.1398289306640625, "cos_r_negg": [0.16477784514427185, 0.23275025188922882, 0.12680009007453918], "cos_innovation_negg": [0.26131999492645264, 0.33194148540496826, 0.6680927872657776], "cos_apical_negg": [0.16477783024311066, 0.23275025188922882, 0.12680009007453918], "cos_Ac_negg": [0.7932016849517822, 0.8220736980438232, 0.7929159998893738], "r_norm": [2.3955013751983643, 2.229098320007324, 2.123361349105835], "g_norm": [0.005031168460845947, 0.005007987841963768, 0.005000755190849304], "traffic_norm": [7.272818565368652, 7.212302207946777, 7.14691162109375], "traffic_residual_norm": [0.2682626247406006, 0.09561178088188171, 2.496471097401809e-06], "traffic_r2": [0.997870683670044, 0.9996808171272278, 1.0]}, "timing": {"warmup_wall_s": 0.4129815101623535, "training_loop_wall_s": 25.69747233390808, "diagnostics_wall_s": 0.0424189567565918, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 25.69747233390808, "evaluation_wall_s": 0.033040761947631836}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 220163584, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p2_t1234_raw_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p2_t1234_raw_d3_s0.json new file mode 100644 index 0000000..f4270a7 --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p2_t1234_raw_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.2, "traffic_seed": 1234, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p2_t1234_raw_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.826, "eval_loss": 1.1236982177734376, "wall_s": 22.603236436843872, "val_acc": 0.826, "val_loss": 1.1236982177734376, "cos_r_negg": [0.14898426830768585, 0.484492689371109, 0.39227524399757385], "cos_innovation_negg": [0.3585495352745056, 0.5124696493148804, 0.8567795753479004], "cos_apical_negg": [0.14898426830768585, 0.484492689371109, 0.39227524399757385], "cos_Ac_negg": [0.6790401339530945, 0.8295716047286987, 0.8522588014602661], "r_norm": [8.034479141235352, 7.94218635559082, 7.859538555145264], "g_norm": [0.004094304982572794, 0.003984258975833654, 0.003940028138458729], "traffic_norm": [7.283234596252441, 7.226955413818359, 7.163242340087891], "traffic_residual_norm": [0.11758914589881897, 0.035705726593732834, 2.679329554666765e-06], "traffic_r2": [0.9993717670440674, 0.9999004006385803, 1.0]}, "timing": {"warmup_wall_s": 0.39117431640625, "training_loop_wall_s": 22.1209397315979, "diagnostics_wall_s": 0.06173276901245117, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 22.1209397315979, "evaluation_wall_s": 0.02790999412536621}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219639296, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p2_t1234_residual_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p2_t1234_residual_d3_s0.json new file mode 100644 index 0000000..db9ae51 --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p2_t1234_residual_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.2, "traffic_seed": 1234, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p2_t1234_residual_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.8096, "eval_loss": 0.9798548706054687, "wall_s": 21.787996768951416, "val_acc": 0.8096, "val_loss": 0.9798548706054687, "cos_r_negg": [0.5013881325721741, 0.6071096062660217, 0.2938528060913086], "cos_innovation_negg": [0.5013881325721741, 0.6071096062660217, 0.2938528060913086], "cos_apical_negg": [0.13821619749069214, 0.18813952803611755, 0.021842097863554955], "cos_Ac_negg": [0.7512470483779907, 0.8237019777297974, 0.4427528977394104], "r_norm": [3.590956449508667, 2.4154646396636963, 2.234744071960449], "g_norm": [0.005231000017374754, 0.005232702940702438, 0.005232943221926689], "traffic_norm": [6.8036627769470215, 6.7608184814453125, 6.640812397003174], "traffic_residual_norm": [1.8223509788513184, 0.28010785579681396, 4.656314558815211e-06], "traffic_r2": [0.9267374873161316, 0.9980725646018982, 1.0]}, "timing": {"warmup_wall_s": 0.4164004325866699, "training_loop_wall_s": 21.277047872543335, "diagnostics_wall_s": 0.06280326843261719, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 21.277047872543335, "evaluation_wall_s": 0.030254840850830078}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219639296, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p2_t1234_taskfit_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p2_t1234_taskfit_d3_s0.json new file mode 100644 index 0000000..7c1e5e2 --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p2_t1234_taskfit_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 0, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.2, "traffic_seed": 1234, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p2_t1234_taskfit_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.8396, "eval_loss": 0.5309772888183594, "wall_s": 19.918225049972534, "val_acc": 0.8396, "val_loss": 0.5309772888183594, "cos_r_negg": [0.4956566393375397, 0.4241202473640442, 0.4146387279033661], "cos_innovation_negg": [0.4956566393375397, 0.4241202473640442, 0.4146387279033661], "cos_apical_negg": [0.29000580310821533, 0.23562070727348328, 0.06952624022960663], "cos_Ac_negg": [0.8628401756286621, 0.7332794070243835, 0.5076884031295776], "r_norm": [1.649580955505371, 2.121540069580078, 2.169412612915039], "g_norm": [0.0025473758578300476, 0.0025481602642685175, 0.0025477693416178226], "traffic_norm": [4.837040901184082, 4.80703067779541, 4.7091264724731445], "traffic_residual_norm": [0.719336986541748, 1.4191416501998901, 1.477724552154541], "traffic_r2": [0.9771323204040527, 0.9129433035850525, 0.9016086459159851]}, "timing": {"warmup_wall_s": 0.4098546504974365, "training_loop_wall_s": 19.387789249420166, "diagnostics_wall_s": 0.07628059387207031, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 19.387789249420166, "evaluation_wall_s": 0.042788028717041016}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219639296, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p2_t5678_matched_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p2_t5678_matched_d3_s0.json new file mode 100644 index 0000000..1c6208a --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p2_t5678_matched_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "match_innovation_norm", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.2, "traffic_seed": 5678, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p2_t5678_matched_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.7664, "eval_loss": 1.719632275390625, "wall_s": 25.001094818115234, "val_acc": 0.7664, "val_loss": 1.719632275390625, "cos_r_negg": [0.05868580937385559, 0.229567751288414, 0.24966660141944885], "cos_innovation_negg": [0.19063562154769897, 0.28243106603622437, 0.6571357250213623], "cos_apical_negg": [0.058685820549726486, 0.2295677363872528, 0.24966660141944885], "cos_Ac_negg": [0.8504146337509155, 0.9249710440635681, 0.8772271275520325], "r_norm": [3.2145090103149414, 2.994147300720215, 2.8843343257904053], "g_norm": [0.005636845715343952, 0.0055962963961064816, 0.005589999258518219], "traffic_norm": [7.43814754486084, 7.493167877197266, 7.290974140167236], "traffic_residual_norm": [0.27292943000793457, 0.11938697844743729, 1.9593508113757707e-06], "traffic_r2": [0.9980160593986511, 0.999626636505127, 1.0]}, "timing": {"warmup_wall_s": 0.4204416275024414, "training_loop_wall_s": 24.509212732315063, "diagnostics_wall_s": 0.03775525093078613, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 24.509212732315063, "evaluation_wall_s": 0.032060861587524414}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 220163584, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p2_t5678_raw_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p2_t5678_raw_d3_s0.json new file mode 100644 index 0000000..a14194c --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p2_t5678_raw_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.2, "traffic_seed": 5678, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p2_t5678_raw_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.7702, "eval_loss": 1.1480270263671875, "wall_s": 22.815719842910767, "val_acc": 0.7702, "val_loss": 1.1480270263671875, "cos_r_negg": [0.06591163575649261, 0.03167945519089699, 0.04799136146903038], "cos_innovation_negg": [0.5248222947120667, 0.6800946593284607, 0.5839487910270691], "cos_apical_negg": [0.06591163575649261, 0.03167945519089699, 0.04799136146903038], "cos_Ac_negg": [0.7705807089805603, 0.7505715489387512, 0.6776537895202637], "r_norm": [8.501500129699707, 8.498258590698242, 8.32878303527832], "g_norm": [0.004900095984339714, 0.004863050766289234, 0.004858908709138632], "traffic_norm": [7.453220367431641, 7.506974220275879, 7.312187194824219], "traffic_residual_norm": [0.13382214307785034, 0.0506257489323616, 2.3795362267264863e-06], "traffic_r2": [0.9993723630905151, 0.9998725056648254, 1.0]}, "timing": {"warmup_wall_s": 0.3735365867614746, "training_loop_wall_s": 22.341991662979126, "diagnostics_wall_s": 0.06650090217590332, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 22.341991662979126, "evaluation_wall_s": 0.032082557678222656}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219639296, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p2_t5678_residual_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p2_t5678_residual_d3_s0.json new file mode 100644 index 0000000..3bf9c48 --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p2_t5678_residual_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.2, "traffic_seed": 5678, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p2_t5678_residual_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.8214, "eval_loss": 0.8180582885742187, "wall_s": 21.150827169418335, "val_acc": 0.8214, "val_loss": 0.8180582885742187, "cos_r_negg": [0.41534242033958435, 0.3307948410511017, 0.6391791105270386], "cos_innovation_negg": [0.41534242033958435, 0.3307948410511017, 0.6391791105270386], "cos_apical_negg": [0.289919912815094, 0.1494867205619812, 0.01514277420938015], "cos_Ac_negg": [0.7311751246452332, 0.7233189344406128, 0.6367807388305664], "r_norm": [3.2990498542785645, 1.6952686309814453, 1.5127090215682983], "g_norm": [0.0035467753186821938, 0.0035455222241580486, 0.003545030951499939], "traffic_norm": [6.962442398071289, 7.110050201416016, 6.900029182434082], "traffic_residual_norm": [2.184058666229248, 0.3132919669151306, 1.3617968761536758e-06], "traffic_r2": [0.8980284333229065, 0.9978842735290527, 1.0]}, "timing": {"warmup_wall_s": 0.34844136238098145, "training_loop_wall_s": 20.700212240219116, "diagnostics_wall_s": 0.07215571403503418, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 20.700212240219116, "evaluation_wall_s": 0.028527021408081055}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219639296, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p2_t5678_taskfit_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p2_t5678_taskfit_d3_s0.json new file mode 100644 index 0000000..2baed9b --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p2_t5678_taskfit_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 0, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.2, "traffic_seed": 5678, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p2_t5678_taskfit_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.746, "eval_loss": 1.157547314453125, "wall_s": 19.882026195526123, "val_acc": 0.746, "val_loss": 1.157547314453125, "cos_r_negg": [0.37024134397506714, 0.41512084007263184, 0.3872508406639099], "cos_innovation_negg": [0.37024134397506714, 0.41512084007263184, 0.3872508406639099], "cos_apical_negg": [0.3464148938655853, -0.05672287940979004, -0.014044417068362236], "cos_Ac_negg": [0.7433834075927734, 0.6700245141983032, 0.5332711935043335], "r_norm": [3.8264167308807373, 2.6155970096588135, 2.539189338684082], "g_norm": [0.005389765370637178, 0.005388968624174595, 0.005387134850025177], "traffic_norm": [6.865685939788818, 6.92400598526001, 6.699646472930908], "traffic_residual_norm": [1.9480180740356445, 0.33052390813827515, 0.2301483452320099], "traffic_r2": [0.9148201942443848, 0.9972731471061707, 0.9988219141960144]}, "timing": {"warmup_wall_s": 0.35330629348754883, "training_loop_wall_s": 19.427300214767456, "diagnostics_wall_s": 0.06629657745361328, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 19.427300214767456, "evaluation_wall_s": 0.03357410430908203}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219639296, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p5_t1234_matched_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p5_t1234_matched_d3_s0.json new file mode 100644 index 0000000..9690e92 --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p5_t1234_matched_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "match_innovation_norm", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 1234, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p5_t1234_matched_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.678, "eval_loss": 3.79026689453125, "wall_s": 23.648794889450073, "val_acc": 0.678, "val_loss": 3.79026689453125, "cos_r_negg": [0.1512756049633026, 0.6518580913543701, 0.21007193624973297], "cos_innovation_negg": [0.3048873245716095, 0.6285126209259033, 0.4548966586589813], "cos_apical_negg": [0.1512756049633026, 0.6518580913543701, 0.21007190644741058], "cos_Ac_negg": [0.7355824112892151, 0.5599015355110168, 0.5938100814819336], "r_norm": [6.336716175079346, 6.008963584899902, 6.106201171875], "g_norm": [0.011333335191011429, 0.011328812688589096, 0.01132880337536335], "traffic_norm": [18.200180053710938, 18.04511260986328, 17.891834259033203], "traffic_residual_norm": [0.3341761827468872, 0.0170009583234787, 9.118590469370247e-07], "traffic_r2": [0.9996628761291504, 0.9999991059303284, 1.0]}, "timing": {"warmup_wall_s": 0.39580416679382324, "training_loop_wall_s": 23.170236825942993, "diagnostics_wall_s": 0.04476046562194824, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 23.170236825942993, "evaluation_wall_s": 0.036089420318603516}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 220163584, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p5_t1234_raw_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p5_t1234_raw_d3_s0.json new file mode 100644 index 0000000..3a7a97d --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p5_t1234_raw_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 1234, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p5_t1234_raw_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.5012, "eval_loss": 6.1343171875, "wall_s": 22.705001831054688, "val_acc": 0.5012, "val_loss": 6.1343171875, "cos_r_negg": [0.2207087129354477, 0.07026983797550201, 0.08709942549467087], "cos_innovation_negg": [0.6479727625846863, 0.5162948966026306, 0.5501437187194824], "cos_apical_negg": [0.2207087129354477, 0.07026983797550201, 0.08709942549467087], "cos_Ac_negg": [0.8656738996505737, 0.681777834892273, 0.5749472379684448], "r_norm": [19.9700927734375, 19.823143005371094, 19.715003967285156], "g_norm": [0.011840133927762508, 0.011817407794296741, 0.01181560754776001], "traffic_norm": [18.207958221435547, 18.060178756713867, 17.885725021362305], "traffic_residual_norm": [0.324695348739624, 0.14401236176490784, 3.5313480566401267e-06], "traffic_r2": [0.9995782375335693, 0.9998794198036194, 1.0]}, "timing": {"warmup_wall_s": 0.41289567947387695, "training_loop_wall_s": 22.19638442993164, "diagnostics_wall_s": 0.06504011154174805, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 22.19638442993164, "evaluation_wall_s": 0.02924633026123047}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219639296, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p5_t1234_residual_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p5_t1234_residual_d3_s0.json new file mode 100644 index 0000000..c6747e1 --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p5_t1234_residual_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 1234, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p5_t1234_residual_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.7514, "eval_loss": 1.31668115234375, "wall_s": 20.48579168319702, "val_acc": 0.7514, "val_loss": 1.31668115234375, "cos_r_negg": [0.4591362476348877, 0.3068358302116394, 0.5801454782485962], "cos_innovation_negg": [0.4591362476348877, 0.3068358302116394, 0.5801454782485962], "cos_apical_negg": [0.13090305030345917, -0.12856262922286987, 0.1256507933139801], "cos_Ac_negg": [0.7498571276664734, 0.7654014825820923, 0.5830812454223633], "r_norm": [2.736917018890381, 2.6897213459014893, 2.2015886306762695], "g_norm": [0.005056717898696661, 0.005056598223745823, 0.005056608468294144], "traffic_norm": [17.072586059570312, 17.01967430114746, 16.970306396484375], "traffic_residual_norm": [0.6206346154212952, 0.6354446411132812, 1.8588855255075032e-06], "traffic_r2": [0.9967383742332458, 0.9984506964683533, 1.0]}, "timing": {"warmup_wall_s": 0.3514277935028076, "training_loop_wall_s": 20.039435386657715, "diagnostics_wall_s": 0.0655221939086914, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 20.039435386657715, "evaluation_wall_s": 0.028061866760253906}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219639296, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p5_t1234_taskfit_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p5_t1234_taskfit_d3_s0.json new file mode 100644 index 0000000..6286945 --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p5_t1234_taskfit_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 0, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 1234, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p5_t1234_taskfit_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.8246, "eval_loss": 0.578564599609375, "wall_s": 20.01433753967285, "val_acc": 0.8246, "val_loss": 0.578564599609375, "cos_r_negg": [0.4286636710166931, 0.35055214166641235, 0.25276824831962585], "cos_innovation_negg": [0.4286636710166931, 0.35055214166641235, 0.25276824831962585], "cos_apical_negg": [0.20668229460716248, 0.07569491118192673, 0.1915556788444519], "cos_Ac_negg": [0.8244516849517822, 0.6607251763343811, 0.6304439306259155], "r_norm": [2.4283409118652344, 2.6578164100646973, 3.942793130874634], "g_norm": [0.0029236176051199436, 0.002904981840401888, 0.00290286005474627], "traffic_norm": [12.157758712768555, 12.307987213134766, 11.931326866149902], "traffic_residual_norm": [1.5154472589492798, 1.9669182300567627, 3.4071197509765625], "traffic_r2": [0.97724848985672, 0.9743850827217102, 0.9185591340065002]}, "timing": {"warmup_wall_s": 0.38341593742370605, "training_loop_wall_s": 19.535274505615234, "diagnostics_wall_s": 0.06669950485229492, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 19.535274505615234, "evaluation_wall_s": 0.02747941017150879}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219639296, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p5_t5678_matched_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p5_t5678_matched_d3_s0.json new file mode 100644 index 0000000..f518a85 --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p5_t5678_matched_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "match_innovation_norm", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 5678, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p5_t5678_matched_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.6704, "eval_loss": 5.4467068359375, "wall_s": 25.0154447555542, "val_acc": 0.6704, "val_loss": 5.4467068359375, "cos_r_negg": [0.33017590641975403, 0.20287546515464783, 0.36235058307647705], "cos_innovation_negg": [0.571456253528595, 0.3596760034561157, 0.6981196999549866], "cos_apical_negg": [0.33017590641975403, 0.20287546515464783, 0.36235061287879944], "cos_Ac_negg": [0.8761051893234253, 0.7164463996887207, 0.778091549873352], "r_norm": [6.855681896209717, 6.510761260986328, 6.530251502990723], "g_norm": [0.01090719923377037, 0.010907039046287537, 0.010907016694545746], "traffic_norm": [18.59433937072754, 18.754268646240234, 18.221567153930664], "traffic_residual_norm": [0.26063138246536255, 0.022629301995038986, 1.246731244464172e-06], "traffic_r2": [0.9998035430908203, 0.9999985694885254, 1.0]}, "timing": {"warmup_wall_s": 0.43842077255249023, "training_loop_wall_s": 24.504605054855347, "diagnostics_wall_s": 0.040040016174316406, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 24.504605054855347, "evaluation_wall_s": 0.030871868133544922}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 220163584, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p5_t5678_raw_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p5_t5678_raw_d3_s0.json new file mode 100644 index 0000000..6a41610 --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p5_t5678_raw_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 0, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 5678, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p5_t5678_raw_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.621, "eval_loss": 3.06246455078125, "wall_s": 21.248085260391235, "val_acc": 0.621, "val_loss": 3.06246455078125, "cos_r_negg": [0.1809769570827484, 0.35078176856040955, 0.1857837587594986], "cos_innovation_negg": [0.2985493540763855, 0.3561640679836273, 0.5450922846794128], "cos_apical_negg": [0.1809769570827484, 0.35078176856040955, 0.1857837587594986], "cos_Ac_negg": [0.7667011618614197, 0.839125394821167, 0.6516420841217041], "r_norm": [19.820140838623047, 19.859249114990234, 19.414657592773438], "g_norm": [0.009958658367395401, 0.009907521307468414, 0.009901592507958412], "traffic_norm": [18.64820671081543, 18.791086196899414, 18.296722412109375], "traffic_residual_norm": [0.07755319029092789, 0.011795603670179844, 2.413739593976061e-06], "traffic_r2": [0.9999665021896362, 0.9999968409538269, 1.0]}, "timing": {"warmup_wall_s": 0.35840678215026855, "training_loop_wall_s": 20.79409694671631, "diagnostics_wall_s": 0.0665292739868164, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 20.79409694671631, "evaluation_wall_s": 0.027467727661132812}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219639296, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p5_t5678_residual_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p5_t5678_residual_d3_s0.json new file mode 100644 index 0000000..bdcfbe0 --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p5_t5678_residual_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 1, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 5678, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p5_t5678_residual_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.7778, "eval_loss": 1.0694345825195313, "wall_s": 21.158701181411743, "val_acc": 0.7778, "val_loss": 1.0694345825195313, "cos_r_negg": [0.42142412066459656, 0.16979441046714783, 0.5140409469604492], "cos_innovation_negg": [0.42142412066459656, 0.16979441046714783, 0.5140409469604492], "cos_apical_negg": [0.08588490635156631, 0.05044504627585411, -0.03644317761063576], "cos_Ac_negg": [0.7878730893135071, 0.4889719486236572, 0.6652572154998779], "r_norm": [2.2845616340637207, 3.299349308013916, 1.8224895000457764], "g_norm": [0.004345351830124855, 0.0043452950194478035, 0.004345091059803963], "traffic_norm": [17.427845001220703, 17.876941680908203, 17.335481643676758], "traffic_residual_norm": [0.5008912086486816, 2.0347583293914795, 1.9694459751917748e-06], "traffic_r2": [0.9971579313278198, 0.986924946308136, 1.0]}, "timing": {"warmup_wall_s": 0.3612501621246338, "training_loop_wall_s": 20.69917345046997, "diagnostics_wall_s": 0.06763291358947754, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 20.69917345046997, "evaluation_wall_s": 0.02924370765686035}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219639296, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file diff --git a/results/c1_fmnist_val_v1_topdown_rho0p5_t5678_taskfit_d3_s0.json b/results/c1_fmnist_val_v1_topdown_rho0p5_t5678_taskfit_d3_s0.json new file mode 100644 index 0000000..38d3f64 --- /dev/null +++ b/results/c1_fmnist_val_v1_topdown_rho0p5_t5678_taskfit_d3_s0.json @@ -0,0 +1 @@ +{"args": {"mode": "sdil", "dataset": "fmnist", "depth": 3, "width": 256, "act": "tanh", "residual": 1, "epochs": 15, "batch_size": 128, "train_examples": 0, "val_examples": 5000, "split_seed": 2027, "eval_split": "validation", "task_seed": 0, "task_train_examples": 50000, "task_test_examples": 10000, "task_levels": 8, "task_n_in": 128, "task_classes": 10, "teacher_depth": 8, "teacher_width": 64, "teacher_residual": 1, "eta": 0.05, "eta_A": 0.02, "eta_P": 0.05, "momentum": 0.9, "w_scale": 1.0, "a_scale": 1.0, "feedback_scale": 1.0, "pert_sigma": 0.01, "pert_every": 4, "pert_ndirs": 1, "pert_mode": "simultaneous", "use_residual": 1, "raw_scale_control": "none", "learn_A": 1, "learn_P": 1, "p_neutral": 0, "p_warmup_steps": 200, "p_warmup_eta": 0.05, "nuis_rho": 0.5, "traffic_seed": 5678, "traffic_mode": "topdown", "predictor_mode": "diagonal", "normalize_delta": 0, "settle_steps": 0, "kappa": 0.0, "feedback": "error", "seed": 0, "device": "cuda", "log_every": 100000, "diagnostics": "alignment", "diagnostics_schedule": "final", "eval_every": 0, "max_steps": 0, "probe_bs": 512, "outdir": "results", "tag": "c1_fmnist_val_v1_topdown_rho0p5_t5678_taskfit_d3_s0", "n_in": 784, "n_out": 10}, "split": {"dataset": "fmnist", "split_seed": 2027, "validation_examples": 5000, "validation_index_sha256": "c5f293471287d93b4149c9ca0f9aec57df1ddb6d499234fcf6fd45f4f80ddf58", "validation_class_counts": {"0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500}, "train_examples": 55000, "test_examples": 10000, "split_from_training_only": true, "evaluation_split": "validation"}, "provenance": {"git_commit": "26407ca3e958ef3701cc69afae4973f023b5b19e", "git_dirty": false}, "diagnostic_protocol": {"probe_source": "training_prefix", "probe_examples": 512, "schedule": "final"}, "steps": [{"step": 0, "epoch": 0, "train_loss": 2.503190279006958}, {"epoch_end": 0, "step": 430}, {"epoch_end": 1, "step": 860}, {"epoch_end": 2, "step": 1290}, {"epoch_end": 3, "step": 1720}, {"epoch_end": 4, "step": 2150}, {"epoch_end": 5, "step": 2580}, {"epoch_end": 6, "step": 3010}, {"epoch_end": 7, "step": 3440}, {"epoch_end": 8, "step": 3870}, {"epoch_end": 9, "step": 4300}, {"epoch_end": 10, "step": 4730}, {"epoch_end": 11, "step": 5160}, {"epoch_end": 12, "step": 5590}, {"epoch_end": 13, "step": 6020}, {"epoch_end": 14, "step": 6450}], "final": {"eval_split": "validation", "eval_acc": 0.828, "eval_loss": 0.6302553955078125, "wall_s": 19.541324853897095, "val_acc": 0.828, "val_loss": 0.6302553955078125, "cos_r_negg": [0.3869021534919739, 0.09084931761026382, 0.3518064618110657], "cos_innovation_negg": [0.3869021534919739, 0.09084931761026382, 0.3518064618110657], "cos_apical_negg": [0.14180129766464233, 0.1204545646905899, -0.018008645623922348], "cos_Ac_negg": [0.6962375640869141, 0.18577462434768677, 0.5927940607070923], "r_norm": [2.2456958293914795, 2.369955539703369, 3.110051155090332], "g_norm": [0.0027911458164453506, 0.0027893483638763428, 0.0027897320687770844], "traffic_norm": [12.074913024902344, 11.838173866271973, 11.718451499938965], "traffic_residual_norm": [1.34244704246521, 1.654012680053711, 2.4815287590026855], "traffic_r2": [0.984882116317749, 0.980353057384491, 0.9551846385002136]}, "timing": {"warmup_wall_s": 0.4113030433654785, "training_loop_wall_s": 19.030887842178345, "diagnostics_wall_s": 0.0661318302154541, "inline_diagnostics_wall_s": 0.0, "optimizer_wall_s_excluding_inline_diagnostics": 19.030887842178345, "evaluation_wall_s": 0.03171992301940918}, "cost": {"train_steps": 6450, "ordinary_training_forward_examples": 825000, "predictor_warmup_forward_examples": 25600, "perturbation_events": 1613, "calibration_batch_loss_evaluations": 3226, "calibration_example_loss_evaluations": 412928, "calibration_forward_equivalent_examples": 619392.0, "training_forward_equivalent_examples": 1469992.0, "max_perturbation_batch_expansion": 2}, "hardware": {"device": "cuda", "torch_version": "2.3.1+cu118", "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "7", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 219639296, "peak_memory_reserved_bytes": 239075328}}
\ No newline at end of file |
