diff options
| author | YurenHao0426 <Blackhao0426@gmail.com> | 2026-07-22 10:56:46 -0500 |
|---|---|---|
| committer | YurenHao0426 <Blackhao0426@gmail.com> | 2026-07-22 10:56:46 -0500 |
| commit | 795d63a4764f6341c8a85471d8cdff310d69f394 (patch) | |
| tree | 250fcab324aa9779aec74015c8d9db79629abf8f /results | |
| parent | 019b274f37a71a70a33d1b651a287ef3d60f026b (diff) | |
results: select Oral-A apical vectorizers
Diffstat (limited to 'results')
13 files changed, 4755 insertions, 0 deletions
diff --git a/results/oral_a_apical_screen/channel_gated_a0_etaA0.001.json b/results/oral_a_apical_screen/channel_gated_a0_etaA0.001.json new file mode 100644 index 0000000..82e71f5 --- /dev/null +++ b/results/oral_a_apical_screen/channel_gated_a0_etaA0.001.json @@ -0,0 +1,378 @@ +{ + "apical_warmup": { + "last": { + "calibration_mse": 518.2959037449049, + "prediction_target_cosine": -2.7486039723463496e-05, + "target_power": 518.2958984375 + }, + "steps": 100 + }, + "architecture": { + "adaptive_apical_parameters": 390592, + "base_width": 16, + "blocks_per_stage": 3, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "depth": 20, + "family": "CIFAR 6n+2 ResNet, option-A shortcuts", + "fixed_traffic_coefficients": 188416, + "forward_parameters": 269722, + "hidden_shapes": [ + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ] + ], + "normalization": "batchnorm", + "predictor_parameters": 376832, + "residual_scale": 1.0, + "vectorizer_mode": "channel_gated", + "vectorizer_parameters": 13760 + }, + "args": { + "a_scale": 0.0, + "a_warmup_steps": 100, + "alignment_probe": 32, + "apical_seed": null, + "augment_train": 1, + "batch_size": 128, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "data_dir": "/home/yurenh2/sdrn/data", + "depth": 20, + "device": "cuda:0", + "epochs": 0, + "eta_A": 0.001, + "eta_P": 0.01, + "eval_every": 0, + "eval_split": "validation", + "learn_P": 0, + "loader_seed": 0, + "lr": 0.03, + "lr_gamma": 0.1, + "lr_milestones": "100,150", + "lr_schedule": "constant", + "max_steps": 0, + "mode": "sdil", + "momentum": 0.9, + "normalization": "batchnorm", + "nuisance_scale": 0.0, + "out": "results/oral_a_apical_screen/channel_gated_a0_etaA0.001.json", + "output_lr": 0.1, + "pert_directions": 1, + "pert_every": 4, + "pert_sigma": 0.01, + "perturb_seed": 1000, + "predictor_warmup_steps": 0, + "residual_scale": null, + "seed": 0, + "split_seed": 2027, + "train_limit": 10000, + "use_residual": 1, + "val_examples": 5000, + "vectorizer_mode": "channel_gated", + "warmup_epochs": 0, + "weight_decay": 0.0001, + "weight_scale": 1.0, + "width": 16 + }, + "counters": { + "apical_warmup_examples": 12688, + "calibration_event_examples": 12688, + "causal_scalar_observations": 200, + "logical_batch_loss_queries": 200, + "ordinary_examples": 0, + "per_example_loss_terms": 25376, + "perturbation_events": 100, + "perturbation_forward_examples": 25376, + "predictor_warmup_examples": 0 + }, + "diagnostics": { + "early_third_mean": -0.0007199425793563327, + "innovation_negative_gradient_cosine": [ + 0.0013892862480133772, + -0.001127532683312893, + 0.0027905036695301533, + 0.001261020777747035, + -0.009130734950304031, + 0.0004978014621883631, + -0.0031196754425764084, + 0.005933355540037155, + -0.021203389391303062, + -0.008318101987242699, + -0.004163672216236591, + 0.018273282796144485, + -0.0047713108360767365, + 0.0017094791401177645, + -0.026866262778639793, + 0.033397890627384186, + -0.01386305596679449, + 0.02709115296602249, + 0.04357969015836716 + ], + "normalization_state": "training_batch_stats_without_running_update", + "raw_negative_gradient_cosine": [ + 0.0013892862480133772, + -0.001127532683312893, + 0.0027905036695301533, + 0.001261020777747035, + -0.009130734950304031, + 0.0004978014621883631, + -0.0031196754425764084, + 0.005933355540037155, + -0.021203389391303062, + -0.008318101987242699, + -0.004163672216236591, + 0.018273282796144485, + -0.0047713108360767365, + 0.0017094791401177645, + -0.026866262778639793, + 0.033397890627384186, + -0.01386305596679449, + 0.02709115296602249, + 0.04357969015836716 + ], + "teaching_negative_gradient_cosine": [ + 0.0013892862480133772, + -0.001127532683312893, + 0.0027905036695301533, + 0.001261020777747035, + -0.009130734950304031, + 0.0004978014621883631, + -0.0031196754425764084, + 0.005933355540037155, + -0.021203389391303062, + -0.008318101987242699, + -0.004163672216236591, + 0.018273282796144485, + -0.0047713108360767365, + 0.0017094791401177645, + -0.026866262778639793, + 0.033397890627384186, + -0.01386305596679449, + 0.02709115296602249, + 0.04357969015836716 + ], + "wall_s": 0.11439704895019531 + }, + "epochs": [], + "evaluation_protocol": { + "test_evaluations": 0, + "test_used_for_selection": false, + "validation_evaluations": 1 + }, + "final": { + "accuracy": 0.1004, + "epoch": 0, + "evaluation_split": "validation", + "finite": true, + "loss": 16.593759375, + "step": 0 + }, + "hardware": { + "cuda_device_name": "NVIDIA GeForce GTX 1080", + "cuda_visible_devices": "5", + "device": "cuda:0", + "device_total_memory_bytes": 8507949056, + "peak_memory_allocated_bytes": 1734868480, + "peak_memory_reserved_bytes": 2690646016, + "torch_version": "2.3.1+cu118" + }, + "protocol_family": "oral_a_cifar_local_resnet_development", + "provenance": { + "git_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "git_tracked_dirty": false + }, + "schema_version": 1, + "split": { + "cifar_source_files": [ + { + "bytes": 31035704, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_1", + "sha256": "54636561a3ce25bd3e19253c6b0d8538147b0ae398331ac4a2d86c6d987368cd" + }, + { + "bytes": 31035320, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_2", + "sha256": "766b2cef9fbc745cf056b3152224f7cf77163b330ea9a15f9392beb8b89bc5a8" + }, + { + "bytes": 31035999, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_3", + "sha256": "0f00d98ebfb30b3ec0ad19f9756dc2630b89003e10525f5e148445e82aa6a1f9" + }, + { + "bytes": 31035696, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_4", + "sha256": "3f7bb240661948b8f4d53e36ec720d8306f5668bd0071dcb4e6c947f78e9682b" + }, + { + "bytes": 31035623, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_5", + "sha256": "d91802434d8376bbaeeadf58a737e3a1b12ac839077e931237e0dcd43adcb154" + }, + { + "bytes": 31035526, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/test_batch", + "sha256": "f53d8d457504f7cff4ea9e021afcf0e0ad8e24a91f3fc42091b8adef61157831" + } + ], + "dataset": "cifar10", + "evaluation_split": "validation", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "random_crop_32_padding_4_zero_then_horizontal_flip_p0.5", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + }, + "timing": { + "apical_warmup_wall_s": 6.52116584777832, + "evaluation_wall_s": 0.37460899353027344, + "predictor_warmup_wall_s": 0.0, + "timing_excludes_data_loading_hashing_and_model_construction": true, + "total_timed_wall_s": 6.895947217941284, + "train_wall_s": 0.0 + }, + "work": { + "apical_macs_per_example": 202176, + "causal_scalar_observations": 200, + "components": { + "apical_projection_macs": 2565209088, + "apical_regression_macs": 2565209088, + "bp_reverse_macs_estimate": 0, + "local_weight_correlation_macs": 0, + "ordinary_forward_macs": 0, + "perturbation_forward_macs": 1029023191040, + "warmup_clean_forward_macs": 514511595520 + }, + "definition": "multiply-accumulates in conv/linear maps; one local weight correlation equals one forward-weight MAC count; BP reverse is estimated as one weight-gradient plus one activation-gradient convolution per forward convolution; elementwise nonlinearities and optimizer arithmetic excluded", + "forward_macs_per_example": 40551040, + "logical_batch_loss_queries": 200, + "per_example_cross_entropy_terms": 25376, + "total_clean_forward_examples": 12688, + "total_forward_equivalent_examples": 38064, + "total_macs_estimate": 1548665204736 + } +} diff --git a/results/oral_a_apical_screen/channel_gated_a0_etaA0.01.json b/results/oral_a_apical_screen/channel_gated_a0_etaA0.01.json new file mode 100644 index 0000000..4a0d734 --- /dev/null +++ b/results/oral_a_apical_screen/channel_gated_a0_etaA0.01.json @@ -0,0 +1,378 @@ +{ + "apical_warmup": { + "last": { + "calibration_mse": 518.2959435504416, + "prediction_target_cosine": -3.4022074583332044e-05, + "target_power": 518.2958984375 + }, + "steps": 100 + }, + "architecture": { + "adaptive_apical_parameters": 390592, + "base_width": 16, + "blocks_per_stage": 3, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "depth": 20, + "family": "CIFAR 6n+2 ResNet, option-A shortcuts", + "fixed_traffic_coefficients": 188416, + "forward_parameters": 269722, + "hidden_shapes": [ + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ] + ], + "normalization": "batchnorm", + "predictor_parameters": 376832, + "residual_scale": 1.0, + "vectorizer_mode": "channel_gated", + "vectorizer_parameters": 13760 + }, + "args": { + "a_scale": 0.0, + "a_warmup_steps": 100, + "alignment_probe": 32, + "apical_seed": null, + "augment_train": 1, + "batch_size": 128, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "data_dir": "/home/yurenh2/sdrn/data", + "depth": 20, + "device": "cuda:0", + "epochs": 0, + "eta_A": 0.01, + "eta_P": 0.01, + "eval_every": 0, + "eval_split": "validation", + "learn_P": 0, + "loader_seed": 0, + "lr": 0.03, + "lr_gamma": 0.1, + "lr_milestones": "100,150", + "lr_schedule": "constant", + "max_steps": 0, + "mode": "sdil", + "momentum": 0.9, + "normalization": "batchnorm", + "nuisance_scale": 0.0, + "out": "results/oral_a_apical_screen/channel_gated_a0_etaA0.01.json", + "output_lr": 0.1, + "pert_directions": 1, + "pert_every": 4, + "pert_sigma": 0.01, + "perturb_seed": 1000, + "predictor_warmup_steps": 0, + "residual_scale": null, + "seed": 0, + "split_seed": 2027, + "train_limit": 10000, + "use_residual": 1, + "val_examples": 5000, + "vectorizer_mode": "channel_gated", + "warmup_epochs": 0, + "weight_decay": 0.0001, + "weight_scale": 1.0, + "width": 16 + }, + "counters": { + "apical_warmup_examples": 12688, + "calibration_event_examples": 12688, + "causal_scalar_observations": 200, + "logical_batch_loss_queries": 200, + "ordinary_examples": 0, + "per_example_loss_terms": 25376, + "perturbation_events": 100, + "perturbation_forward_examples": 25376, + "predictor_warmup_examples": 0 + }, + "diagnostics": { + "early_third_mean": -0.0006589657083774606, + "innovation_negative_gradient_cosine": [ + 0.0013765119947493076, + -0.0013477427419275045, + 0.0029596295207738876, + 0.0013037732569500804, + -0.009216060861945152, + 0.0009700945811346173, + -0.004138186573982239, + 0.005998332519084215, + -0.021400559693574905, + -0.00809585303068161, + -0.0038684492465108633, + 0.018366698175668716, + -0.003738394705578685, + 0.0025836736895143986, + -0.02535463497042656, + 0.03396084904670715, + -0.016801325604319572, + 0.028639018535614014, + 0.04321654140949249 + ], + "normalization_state": "training_batch_stats_without_running_update", + "raw_negative_gradient_cosine": [ + 0.0013765119947493076, + -0.0013477427419275045, + 0.0029596295207738876, + 0.0013037732569500804, + -0.009216060861945152, + 0.0009700945811346173, + -0.004138186573982239, + 0.005998332519084215, + -0.021400559693574905, + -0.00809585303068161, + -0.0038684492465108633, + 0.018366698175668716, + -0.003738394705578685, + 0.0025836736895143986, + -0.02535463497042656, + 0.03396084904670715, + -0.016801325604319572, + 0.028639018535614014, + 0.04321654140949249 + ], + "teaching_negative_gradient_cosine": [ + 0.0013765119947493076, + -0.0013477427419275045, + 0.0029596295207738876, + 0.0013037732569500804, + -0.009216060861945152, + 0.0009700945811346173, + -0.004138186573982239, + 0.005998332519084215, + -0.021400559693574905, + -0.00809585303068161, + -0.0038684492465108633, + 0.018366698175668716, + -0.003738394705578685, + 0.0025836736895143986, + -0.02535463497042656, + 0.03396084904670715, + -0.016801325604319572, + 0.028639018535614014, + 0.04321654140949249 + ], + "wall_s": 0.10339760780334473 + }, + "epochs": [], + "evaluation_protocol": { + "test_evaluations": 0, + "test_used_for_selection": false, + "validation_evaluations": 1 + }, + "final": { + "accuracy": 0.1004, + "epoch": 0, + "evaluation_split": "validation", + "finite": true, + "loss": 16.593759375, + "step": 0 + }, + "hardware": { + "cuda_device_name": "NVIDIA GeForce GTX 1080", + "cuda_visible_devices": "5", + "device": "cuda:0", + "device_total_memory_bytes": 8507949056, + "peak_memory_allocated_bytes": 1734868480, + "peak_memory_reserved_bytes": 2690646016, + "torch_version": "2.3.1+cu118" + }, + "protocol_family": "oral_a_cifar_local_resnet_development", + "provenance": { + "git_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "git_tracked_dirty": false + }, + "schema_version": 1, + "split": { + "cifar_source_files": [ + { + "bytes": 31035704, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_1", + "sha256": "54636561a3ce25bd3e19253c6b0d8538147b0ae398331ac4a2d86c6d987368cd" + }, + { + "bytes": 31035320, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_2", + "sha256": "766b2cef9fbc745cf056b3152224f7cf77163b330ea9a15f9392beb8b89bc5a8" + }, + { + "bytes": 31035999, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_3", + "sha256": "0f00d98ebfb30b3ec0ad19f9756dc2630b89003e10525f5e148445e82aa6a1f9" + }, + { + "bytes": 31035696, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_4", + "sha256": "3f7bb240661948b8f4d53e36ec720d8306f5668bd0071dcb4e6c947f78e9682b" + }, + { + "bytes": 31035623, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_5", + "sha256": "d91802434d8376bbaeeadf58a737e3a1b12ac839077e931237e0dcd43adcb154" + }, + { + "bytes": 31035526, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/test_batch", + "sha256": "f53d8d457504f7cff4ea9e021afcf0e0ad8e24a91f3fc42091b8adef61157831" + } + ], + "dataset": "cifar10", + "evaluation_split": "validation", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "random_crop_32_padding_4_zero_then_horizontal_flip_p0.5", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + }, + "timing": { + "apical_warmup_wall_s": 6.441373586654663, + "evaluation_wall_s": 0.39517831802368164, + "predictor_warmup_wall_s": 0.0, + "timing_excludes_data_loading_hashing_and_model_construction": true, + "total_timed_wall_s": 6.836787462234497, + "train_wall_s": 0.0 + }, + "work": { + "apical_macs_per_example": 202176, + "causal_scalar_observations": 200, + "components": { + "apical_projection_macs": 2565209088, + "apical_regression_macs": 2565209088, + "bp_reverse_macs_estimate": 0, + "local_weight_correlation_macs": 0, + "ordinary_forward_macs": 0, + "perturbation_forward_macs": 1029023191040, + "warmup_clean_forward_macs": 514511595520 + }, + "definition": "multiply-accumulates in conv/linear maps; one local weight correlation equals one forward-weight MAC count; BP reverse is estimated as one weight-gradient plus one activation-gradient convolution per forward convolution; elementwise nonlinearities and optimizer arithmetic excluded", + "forward_macs_per_example": 40551040, + "logical_batch_loss_queries": 200, + "per_example_cross_entropy_terms": 25376, + "total_clean_forward_examples": 12688, + "total_forward_equivalent_examples": 38064, + "total_macs_estimate": 1548665204736 + } +} diff --git a/results/oral_a_apical_screen/channel_gated_a0_etaA0.1.json b/results/oral_a_apical_screen/channel_gated_a0_etaA0.1.json new file mode 100644 index 0000000..1e7ceed --- /dev/null +++ b/results/oral_a_apical_screen/channel_gated_a0_etaA0.1.json @@ -0,0 +1,378 @@ +{ + "apical_warmup": { + "last": { + "calibration_mse": 518.2965393066406, + "prediction_target_cosine": -7.547625140869704e-05, + "target_power": 518.2958984375 + }, + "steps": 100 + }, + "architecture": { + "adaptive_apical_parameters": 390592, + "base_width": 16, + "blocks_per_stage": 3, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "depth": 20, + "family": "CIFAR 6n+2 ResNet, option-A shortcuts", + "fixed_traffic_coefficients": 188416, + "forward_parameters": 269722, + "hidden_shapes": [ + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ] + ], + "normalization": "batchnorm", + "predictor_parameters": 376832, + "residual_scale": 1.0, + "vectorizer_mode": "channel_gated", + "vectorizer_parameters": 13760 + }, + "args": { + "a_scale": 0.0, + "a_warmup_steps": 100, + "alignment_probe": 32, + "apical_seed": null, + "augment_train": 1, + "batch_size": 128, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "data_dir": "/home/yurenh2/sdrn/data", + "depth": 20, + "device": "cuda:0", + "epochs": 0, + "eta_A": 0.1, + "eta_P": 0.01, + "eval_every": 0, + "eval_split": "validation", + "learn_P": 0, + "loader_seed": 0, + "lr": 0.03, + "lr_gamma": 0.1, + "lr_milestones": "100,150", + "lr_schedule": "constant", + "max_steps": 0, + "mode": "sdil", + "momentum": 0.9, + "normalization": "batchnorm", + "nuisance_scale": 0.0, + "out": "results/oral_a_apical_screen/channel_gated_a0_etaA0.1.json", + "output_lr": 0.1, + "pert_directions": 1, + "pert_every": 4, + "pert_sigma": 0.01, + "perturb_seed": 1000, + "predictor_warmup_steps": 0, + "residual_scale": null, + "seed": 0, + "split_seed": 2027, + "train_limit": 10000, + "use_residual": 1, + "val_examples": 5000, + "vectorizer_mode": "channel_gated", + "warmup_epochs": 0, + "weight_decay": 0.0001, + "weight_scale": 1.0, + "width": 16 + }, + "counters": { + "apical_warmup_examples": 12688, + "calibration_event_examples": 12688, + "causal_scalar_observations": 200, + "logical_batch_loss_queries": 200, + "ordinary_examples": 0, + "per_example_loss_terms": 25376, + "perturbation_events": 100, + "perturbation_forward_examples": 25376, + "predictor_warmup_examples": 0 + }, + "diagnostics": { + "early_third_mean": -9.475171100348234e-05, + "innovation_negative_gradient_cosine": [ + 0.0015504095936194062, + -0.0027303805109113455, + 0.004028514958918095, + 0.001691187615506351, + -0.009150981903076172, + 0.0040427399799227715, + -0.010961052030324936, + 0.0058936444111168385, + -0.021497301757335663, + -0.005660664290189743, + -0.0022220034152269363, + 0.018607381731271744, + 0.0069678472355008125, + 0.00933452695608139, + -0.019126977771520615, + 0.03700929880142212, + -0.03835133835673332, + 0.03671898692846298, + 0.03983813896775246 + ], + "normalization_state": "training_batch_stats_without_running_update", + "raw_negative_gradient_cosine": [ + 0.0015504095936194062, + -0.0027303805109113455, + 0.004028514958918095, + 0.001691187615506351, + -0.009150981903076172, + 0.0040427399799227715, + -0.010961052030324936, + 0.0058936444111168385, + -0.021497301757335663, + -0.005660664290189743, + -0.0022220034152269363, + 0.018607381731271744, + 0.0069678472355008125, + 0.00933452695608139, + -0.019126977771520615, + 0.03700929880142212, + -0.03835133835673332, + 0.03671898692846298, + 0.03983813896775246 + ], + "teaching_negative_gradient_cosine": [ + 0.0015504095936194062, + -0.0027303805109113455, + 0.004028514958918095, + 0.001691187615506351, + -0.009150981903076172, + 0.0040427399799227715, + -0.010961052030324936, + 0.0058936444111168385, + -0.021497301757335663, + -0.005660664290189743, + -0.0022220034152269363, + 0.018607381731271744, + 0.0069678472355008125, + 0.00933452695608139, + -0.019126977771520615, + 0.03700929880142212, + -0.03835133835673332, + 0.03671898692846298, + 0.03983813896775246 + ], + "wall_s": 0.12019467353820801 + }, + "epochs": [], + "evaluation_protocol": { + "test_evaluations": 0, + "test_used_for_selection": false, + "validation_evaluations": 1 + }, + "final": { + "accuracy": 0.1004, + "epoch": 0, + "evaluation_split": "validation", + "finite": true, + "loss": 16.593759375, + "step": 0 + }, + "hardware": { + "cuda_device_name": "NVIDIA GeForce GTX 1080", + "cuda_visible_devices": "5", + "device": "cuda:0", + "device_total_memory_bytes": 8507949056, + "peak_memory_allocated_bytes": 1734868480, + "peak_memory_reserved_bytes": 2690646016, + "torch_version": "2.3.1+cu118" + }, + "protocol_family": "oral_a_cifar_local_resnet_development", + "provenance": { + "git_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "git_tracked_dirty": false + }, + "schema_version": 1, + "split": { + "cifar_source_files": [ + { + "bytes": 31035704, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_1", + "sha256": "54636561a3ce25bd3e19253c6b0d8538147b0ae398331ac4a2d86c6d987368cd" + }, + { + "bytes": 31035320, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_2", + "sha256": "766b2cef9fbc745cf056b3152224f7cf77163b330ea9a15f9392beb8b89bc5a8" + }, + { + "bytes": 31035999, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_3", + "sha256": "0f00d98ebfb30b3ec0ad19f9756dc2630b89003e10525f5e148445e82aa6a1f9" + }, + { + "bytes": 31035696, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_4", + "sha256": "3f7bb240661948b8f4d53e36ec720d8306f5668bd0071dcb4e6c947f78e9682b" + }, + { + "bytes": 31035623, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_5", + "sha256": "d91802434d8376bbaeeadf58a737e3a1b12ac839077e931237e0dcd43adcb154" + }, + { + "bytes": 31035526, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/test_batch", + "sha256": "f53d8d457504f7cff4ea9e021afcf0e0ad8e24a91f3fc42091b8adef61157831" + } + ], + "dataset": "cifar10", + "evaluation_split": "validation", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "random_crop_32_padding_4_zero_then_horizontal_flip_p0.5", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + }, + "timing": { + "apical_warmup_wall_s": 6.3952882289886475, + "evaluation_wall_s": 0.37603163719177246, + "predictor_warmup_wall_s": 0.0, + "timing_excludes_data_loading_hashing_and_model_construction": true, + "total_timed_wall_s": 6.773483753204346, + "train_wall_s": 0.0 + }, + "work": { + "apical_macs_per_example": 202176, + "causal_scalar_observations": 200, + "components": { + "apical_projection_macs": 2565209088, + "apical_regression_macs": 2565209088, + "bp_reverse_macs_estimate": 0, + "local_weight_correlation_macs": 0, + "ordinary_forward_macs": 0, + "perturbation_forward_macs": 1029023191040, + "warmup_clean_forward_macs": 514511595520 + }, + "definition": "multiply-accumulates in conv/linear maps; one local weight correlation equals one forward-weight MAC count; BP reverse is estimated as one weight-gradient plus one activation-gradient convolution per forward convolution; elementwise nonlinearities and optimizer arithmetic excluded", + "forward_macs_per_example": 40551040, + "logical_batch_loss_queries": 200, + "per_example_cross_entropy_terms": 25376, + "total_clean_forward_examples": 12688, + "total_forward_equivalent_examples": 38064, + "total_macs_estimate": 1548665204736 + } +} diff --git a/results/oral_a_apical_screen/channel_gated_a1_etaA0.001.json b/results/oral_a_apical_screen/channel_gated_a1_etaA0.001.json new file mode 100644 index 0000000..021d8f4 --- /dev/null +++ b/results/oral_a_apical_screen/channel_gated_a1_etaA0.001.json @@ -0,0 +1,378 @@ +{ + "apical_warmup": { + "last": { + "calibration_mse": 518.2959050717561, + "prediction_target_cosine": 0.00016553175127441687, + "target_power": 518.2958984375 + }, + "steps": 100 + }, + "architecture": { + "adaptive_apical_parameters": 390592, + "base_width": 16, + "blocks_per_stage": 3, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "depth": 20, + "family": "CIFAR 6n+2 ResNet, option-A shortcuts", + "fixed_traffic_coefficients": 188416, + "forward_parameters": 269722, + "hidden_shapes": [ + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ] + ], + "normalization": "batchnorm", + "predictor_parameters": 376832, + "residual_scale": 1.0, + "vectorizer_mode": "channel_gated", + "vectorizer_parameters": 13760 + }, + "args": { + "a_scale": 1.0, + "a_warmup_steps": 100, + "alignment_probe": 32, + "apical_seed": null, + "augment_train": 1, + "batch_size": 128, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "data_dir": "/home/yurenh2/sdrn/data", + "depth": 20, + "device": "cuda:0", + "epochs": 0, + "eta_A": 0.001, + "eta_P": 0.01, + "eval_every": 0, + "eval_split": "validation", + "learn_P": 0, + "loader_seed": 0, + "lr": 0.03, + "lr_gamma": 0.1, + "lr_milestones": "100,150", + "lr_schedule": "constant", + "max_steps": 0, + "mode": "sdil", + "momentum": 0.9, + "normalization": "batchnorm", + "nuisance_scale": 0.0, + "out": "results/oral_a_apical_screen/channel_gated_a1_etaA0.001.json", + "output_lr": 0.1, + "pert_directions": 1, + "pert_every": 4, + "pert_sigma": 0.01, + "perturb_seed": 1000, + "predictor_warmup_steps": 0, + "residual_scale": null, + "seed": 0, + "split_seed": 2027, + "train_limit": 10000, + "use_residual": 1, + "val_examples": 5000, + "vectorizer_mode": "channel_gated", + "warmup_epochs": 0, + "weight_decay": 0.0001, + "weight_scale": 1.0, + "width": 16 + }, + "counters": { + "apical_warmup_examples": 12688, + "calibration_event_examples": 12688, + "causal_scalar_observations": 200, + "logical_batch_loss_queries": 200, + "ordinary_examples": 0, + "per_example_loss_terms": 25376, + "perturbation_events": 100, + "perturbation_forward_examples": 25376, + "predictor_warmup_examples": 0 + }, + "diagnostics": { + "early_third_mean": 0.0014168331302547206, + "innovation_negative_gradient_cosine": [ + 0.0023974194191396236, + -0.0007454301812686026, + 0.004026529844850302, + 0.007125279866158962, + -0.004734348971396685, + 0.0004315488040447235, + -0.00734263751655817, + 0.01816391944885254, + -0.00493968091905117, + -0.009824810549616814, + -0.017024271190166473, + -0.0031740786507725716, + -0.0011432988103479147, + 0.007039900869131088, + -0.026091966778039932, + 0.004069922957569361, + -0.02082895115017891, + 0.00648832693696022, + -0.014036145992577076 + ], + "normalization_state": "training_batch_stats_without_running_update", + "raw_negative_gradient_cosine": [ + 0.0023974194191396236, + -0.0007454301812686026, + 0.004026529844850302, + 0.007125279866158962, + -0.004734348971396685, + 0.0004315488040447235, + -0.00734263751655817, + 0.01816391944885254, + -0.00493968091905117, + -0.009824810549616814, + -0.017024271190166473, + -0.0031740786507725716, + -0.0011432988103479147, + 0.007039900869131088, + -0.026091966778039932, + 0.004069922957569361, + -0.02082895115017891, + 0.00648832693696022, + -0.014036145992577076 + ], + "teaching_negative_gradient_cosine": [ + 0.0023974194191396236, + -0.0007454301812686026, + 0.004026529844850302, + 0.007125279866158962, + -0.004734348971396685, + 0.0004315488040447235, + -0.00734263751655817, + 0.01816391944885254, + -0.00493968091905117, + -0.009824810549616814, + -0.017024271190166473, + -0.0031740786507725716, + -0.0011432988103479147, + 0.007039900869131088, + -0.026091966778039932, + 0.004069922957569361, + -0.02082895115017891, + 0.00648832693696022, + -0.014036145992577076 + ], + "wall_s": 0.11137032508850098 + }, + "epochs": [], + "evaluation_protocol": { + "test_evaluations": 0, + "test_used_for_selection": false, + "validation_evaluations": 1 + }, + "final": { + "accuracy": 0.1004, + "epoch": 0, + "evaluation_split": "validation", + "finite": true, + "loss": 16.593759375, + "step": 0 + }, + "hardware": { + "cuda_device_name": "NVIDIA GeForce GTX 1080", + "cuda_visible_devices": "5", + "device": "cuda:0", + "device_total_memory_bytes": 8507949056, + "peak_memory_allocated_bytes": 1734868480, + "peak_memory_reserved_bytes": 2690646016, + "torch_version": "2.3.1+cu118" + }, + "protocol_family": "oral_a_cifar_local_resnet_development", + "provenance": { + "git_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "git_tracked_dirty": false + }, + "schema_version": 1, + "split": { + "cifar_source_files": [ + { + "bytes": 31035704, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_1", + "sha256": "54636561a3ce25bd3e19253c6b0d8538147b0ae398331ac4a2d86c6d987368cd" + }, + { + "bytes": 31035320, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_2", + "sha256": "766b2cef9fbc745cf056b3152224f7cf77163b330ea9a15f9392beb8b89bc5a8" + }, + { + "bytes": 31035999, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_3", + "sha256": "0f00d98ebfb30b3ec0ad19f9756dc2630b89003e10525f5e148445e82aa6a1f9" + }, + { + "bytes": 31035696, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_4", + "sha256": "3f7bb240661948b8f4d53e36ec720d8306f5668bd0071dcb4e6c947f78e9682b" + }, + { + "bytes": 31035623, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_5", + "sha256": "d91802434d8376bbaeeadf58a737e3a1b12ac839077e931237e0dcd43adcb154" + }, + { + "bytes": 31035526, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/test_batch", + "sha256": "f53d8d457504f7cff4ea9e021afcf0e0ad8e24a91f3fc42091b8adef61157831" + } + ], + "dataset": "cifar10", + "evaluation_split": "validation", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "random_crop_32_padding_4_zero_then_horizontal_flip_p0.5", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + }, + "timing": { + "apical_warmup_wall_s": 6.463217735290527, + "evaluation_wall_s": 0.3811209201812744, + "predictor_warmup_wall_s": 0.0, + "timing_excludes_data_loading_hashing_and_model_construction": true, + "total_timed_wall_s": 6.844555139541626, + "train_wall_s": 0.0 + }, + "work": { + "apical_macs_per_example": 202176, + "causal_scalar_observations": 200, + "components": { + "apical_projection_macs": 2565209088, + "apical_regression_macs": 2565209088, + "bp_reverse_macs_estimate": 0, + "local_weight_correlation_macs": 0, + "ordinary_forward_macs": 0, + "perturbation_forward_macs": 1029023191040, + "warmup_clean_forward_macs": 514511595520 + }, + "definition": "multiply-accumulates in conv/linear maps; one local weight correlation equals one forward-weight MAC count; BP reverse is estimated as one weight-gradient plus one activation-gradient convolution per forward convolution; elementwise nonlinearities and optimizer arithmetic excluded", + "forward_macs_per_example": 40551040, + "logical_batch_loss_queries": 200, + "per_example_cross_entropy_terms": 25376, + "total_clean_forward_examples": 12688, + "total_forward_equivalent_examples": 38064, + "total_macs_estimate": 1548665204736 + } +} diff --git a/results/oral_a_apical_screen/channel_gated_a1_etaA0.01.json b/results/oral_a_apical_screen/channel_gated_a1_etaA0.01.json new file mode 100644 index 0000000..55bbc55 --- /dev/null +++ b/results/oral_a_apical_screen/channel_gated_a1_etaA0.01.json @@ -0,0 +1,378 @@ +{ + "apical_warmup": { + "last": { + "calibration_mse": 518.2959276282269, + "prediction_target_cosine": 3.61075533430761e-05, + "target_power": 518.2958984375 + }, + "steps": 100 + }, + "architecture": { + "adaptive_apical_parameters": 390592, + "base_width": 16, + "blocks_per_stage": 3, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "depth": 20, + "family": "CIFAR 6n+2 ResNet, option-A shortcuts", + "fixed_traffic_coefficients": 188416, + "forward_parameters": 269722, + "hidden_shapes": [ + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ] + ], + "normalization": "batchnorm", + "predictor_parameters": 376832, + "residual_scale": 1.0, + "vectorizer_mode": "channel_gated", + "vectorizer_parameters": 13760 + }, + "args": { + "a_scale": 1.0, + "a_warmup_steps": 100, + "alignment_probe": 32, + "apical_seed": null, + "augment_train": 1, + "batch_size": 128, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "data_dir": "/home/yurenh2/sdrn/data", + "depth": 20, + "device": "cuda:0", + "epochs": 0, + "eta_A": 0.01, + "eta_P": 0.01, + "eval_every": 0, + "eval_split": "validation", + "learn_P": 0, + "loader_seed": 0, + "lr": 0.03, + "lr_gamma": 0.1, + "lr_milestones": "100,150", + "lr_schedule": "constant", + "max_steps": 0, + "mode": "sdil", + "momentum": 0.9, + "normalization": "batchnorm", + "nuisance_scale": 0.0, + "out": "results/oral_a_apical_screen/channel_gated_a1_etaA0.01.json", + "output_lr": 0.1, + "pert_directions": 1, + "pert_every": 4, + "pert_sigma": 0.01, + "perturb_seed": 1000, + "predictor_warmup_steps": 0, + "residual_scale": null, + "seed": 0, + "split_seed": 2027, + "train_limit": 10000, + "use_residual": 1, + "val_examples": 5000, + "vectorizer_mode": "channel_gated", + "warmup_epochs": 0, + "weight_decay": 0.0001, + "weight_scale": 1.0, + "width": 16 + }, + "counters": { + "apical_warmup_examples": 12688, + "calibration_event_examples": 12688, + "causal_scalar_observations": 200, + "logical_batch_loss_queries": 200, + "ordinary_examples": 0, + "per_example_loss_terms": 25376, + "perturbation_events": 100, + "perturbation_forward_examples": 25376, + "predictor_warmup_examples": 0 + }, + "diagnostics": { + "early_third_mean": -0.00027415525012960035, + "innovation_negative_gradient_cosine": [ + 0.0017035005148500204, + -0.001508736633695662, + 0.0034171901643276215, + 0.00247080042026937, + -0.008795436471700668, + 0.0010677505051717162, + -0.005078752525150776, + 0.010226677171885967, + -0.019980795681476593, + -0.009829400107264519, + -0.008519536815583706, + 0.015604992397129536, + -0.0029746461659669876, + 0.005058863200247288, + -0.03246579319238663, + 0.029367517679929733, + -0.02012300305068493, + 0.025693893432617188, + 0.024077557027339935 + ], + "normalization_state": "training_batch_stats_without_running_update", + "raw_negative_gradient_cosine": [ + 0.0017035005148500204, + -0.001508736633695662, + 0.0034171901643276215, + 0.00247080042026937, + -0.008795436471700668, + 0.0010677505051717162, + -0.005078752525150776, + 0.010226677171885967, + -0.019980795681476593, + -0.009829400107264519, + -0.008519536815583706, + 0.015604992397129536, + -0.0029746461659669876, + 0.005058863200247288, + -0.03246579319238663, + 0.029367517679929733, + -0.02012300305068493, + 0.025693893432617188, + 0.024077557027339935 + ], + "teaching_negative_gradient_cosine": [ + 0.0017035005148500204, + -0.001508736633695662, + 0.0034171901643276215, + 0.00247080042026937, + -0.008795436471700668, + 0.0010677505051717162, + -0.005078752525150776, + 0.010226677171885967, + -0.019980795681476593, + -0.009829400107264519, + -0.008519536815583706, + 0.015604992397129536, + -0.0029746461659669876, + 0.005058863200247288, + -0.03246579319238663, + 0.029367517679929733, + -0.02012300305068493, + 0.025693893432617188, + 0.024077557027339935 + ], + "wall_s": 0.09987139701843262 + }, + "epochs": [], + "evaluation_protocol": { + "test_evaluations": 0, + "test_used_for_selection": false, + "validation_evaluations": 1 + }, + "final": { + "accuracy": 0.1004, + "epoch": 0, + "evaluation_split": "validation", + "finite": true, + "loss": 16.593759375, + "step": 0 + }, + "hardware": { + "cuda_device_name": "NVIDIA GeForce GTX 1080", + "cuda_visible_devices": "5", + "device": "cuda:0", + "device_total_memory_bytes": 8507949056, + "peak_memory_allocated_bytes": 1734868480, + "peak_memory_reserved_bytes": 2690646016, + "torch_version": "2.3.1+cu118" + }, + "protocol_family": "oral_a_cifar_local_resnet_development", + "provenance": { + "git_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "git_tracked_dirty": false + }, + "schema_version": 1, + "split": { + "cifar_source_files": [ + { + "bytes": 31035704, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_1", + "sha256": "54636561a3ce25bd3e19253c6b0d8538147b0ae398331ac4a2d86c6d987368cd" + }, + { + "bytes": 31035320, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_2", + "sha256": "766b2cef9fbc745cf056b3152224f7cf77163b330ea9a15f9392beb8b89bc5a8" + }, + { + "bytes": 31035999, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_3", + "sha256": "0f00d98ebfb30b3ec0ad19f9756dc2630b89003e10525f5e148445e82aa6a1f9" + }, + { + "bytes": 31035696, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_4", + "sha256": "3f7bb240661948b8f4d53e36ec720d8306f5668bd0071dcb4e6c947f78e9682b" + }, + { + "bytes": 31035623, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_5", + "sha256": "d91802434d8376bbaeeadf58a737e3a1b12ac839077e931237e0dcd43adcb154" + }, + { + "bytes": 31035526, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/test_batch", + "sha256": "f53d8d457504f7cff4ea9e021afcf0e0ad8e24a91f3fc42091b8adef61157831" + } + ], + "dataset": "cifar10", + "evaluation_split": "validation", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "random_crop_32_padding_4_zero_then_horizontal_flip_p0.5", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + }, + "timing": { + "apical_warmup_wall_s": 6.459696054458618, + "evaluation_wall_s": 0.3754451274871826, + "predictor_warmup_wall_s": 0.0, + "timing_excludes_data_loading_hashing_and_model_construction": true, + "total_timed_wall_s": 6.835346221923828, + "train_wall_s": 0.0 + }, + "work": { + "apical_macs_per_example": 202176, + "causal_scalar_observations": 200, + "components": { + "apical_projection_macs": 2565209088, + "apical_regression_macs": 2565209088, + "bp_reverse_macs_estimate": 0, + "local_weight_correlation_macs": 0, + "ordinary_forward_macs": 0, + "perturbation_forward_macs": 1029023191040, + "warmup_clean_forward_macs": 514511595520 + }, + "definition": "multiply-accumulates in conv/linear maps; one local weight correlation equals one forward-weight MAC count; BP reverse is estimated as one weight-gradient plus one activation-gradient convolution per forward convolution; elementwise nonlinearities and optimizer arithmetic excluded", + "forward_macs_per_example": 40551040, + "logical_batch_loss_queries": 200, + "per_example_cross_entropy_terms": 25376, + "total_clean_forward_examples": 12688, + "total_forward_equivalent_examples": 38064, + "total_macs_estimate": 1548665204736 + } +} diff --git a/results/oral_a_apical_screen/channel_gated_a1_etaA0.1.json b/results/oral_a_apical_screen/channel_gated_a1_etaA0.1.json new file mode 100644 index 0000000..55ee4e0 --- /dev/null +++ b/results/oral_a_apical_screen/channel_gated_a1_etaA0.1.json @@ -0,0 +1,378 @@ +{ + "apical_warmup": { + "last": { + "calibration_mse": 518.296547267748, + "prediction_target_cosine": -7.364532238643221e-05, + "target_power": 518.2958984375 + }, + "steps": 100 + }, + "architecture": { + "adaptive_apical_parameters": 390592, + "base_width": 16, + "blocks_per_stage": 3, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "depth": 20, + "family": "CIFAR 6n+2 ResNet, option-A shortcuts", + "fixed_traffic_coefficients": 188416, + "forward_parameters": 269722, + "hidden_shapes": [ + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ] + ], + "normalization": "batchnorm", + "predictor_parameters": 376832, + "residual_scale": 1.0, + "vectorizer_mode": "channel_gated", + "vectorizer_parameters": 13760 + }, + "args": { + "a_scale": 1.0, + "a_warmup_steps": 100, + "alignment_probe": 32, + "apical_seed": null, + "augment_train": 1, + "batch_size": 128, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "data_dir": "/home/yurenh2/sdrn/data", + "depth": 20, + "device": "cuda:0", + "epochs": 0, + "eta_A": 0.1, + "eta_P": 0.01, + "eval_every": 0, + "eval_split": "validation", + "learn_P": 0, + "loader_seed": 0, + "lr": 0.03, + "lr_gamma": 0.1, + "lr_milestones": "100,150", + "lr_schedule": "constant", + "max_steps": 0, + "mode": "sdil", + "momentum": 0.9, + "normalization": "batchnorm", + "nuisance_scale": 0.0, + "out": "results/oral_a_apical_screen/channel_gated_a1_etaA0.1.json", + "output_lr": 0.1, + "pert_directions": 1, + "pert_every": 4, + "pert_sigma": 0.01, + "perturb_seed": 1000, + "predictor_warmup_steps": 0, + "residual_scale": null, + "seed": 0, + "split_seed": 2027, + "train_limit": 10000, + "use_residual": 1, + "val_examples": 5000, + "vectorizer_mode": "channel_gated", + "warmup_epochs": 0, + "weight_decay": 0.0001, + "weight_scale": 1.0, + "width": 16 + }, + "counters": { + "apical_warmup_examples": 12688, + "calibration_event_examples": 12688, + "causal_scalar_observations": 200, + "logical_batch_loss_queries": 200, + "ordinary_examples": 0, + "per_example_loss_terms": 25376, + "perturbation_events": 100, + "perturbation_forward_examples": 25376, + "predictor_warmup_examples": 0 + }, + "diagnostics": { + "early_third_mean": -7.813629539062579e-05, + "innovation_negative_gradient_cosine": [ + 0.001576898037455976, + -0.0027433838695287704, + 0.00403923774138093, + 0.0017451172461733222, + -0.009138381108641624, + 0.004051694180816412, + -0.011002816259860992, + 0.0061003053560853004, + -0.021391529589891434, + -0.005690492689609528, + -0.002404250204563141, + 0.018496904522180557, + 0.006963520310819149, + 0.00920878816395998, + -0.019383680075407028, + 0.037107981741428375, + -0.038281023502349854, + 0.036819469183683395, + 0.038529884070158005 + ], + "normalization_state": "training_batch_stats_without_running_update", + "raw_negative_gradient_cosine": [ + 0.001576898037455976, + -0.0027433838695287704, + 0.00403923774138093, + 0.0017451172461733222, + -0.009138381108641624, + 0.004051694180816412, + -0.011002816259860992, + 0.0061003053560853004, + -0.021391529589891434, + -0.005690492689609528, + -0.002404250204563141, + 0.018496904522180557, + 0.006963520310819149, + 0.00920878816395998, + -0.019383680075407028, + 0.037107981741428375, + -0.038281023502349854, + 0.036819469183683395, + 0.038529884070158005 + ], + "teaching_negative_gradient_cosine": [ + 0.001576898037455976, + -0.0027433838695287704, + 0.00403923774138093, + 0.0017451172461733222, + -0.009138381108641624, + 0.004051694180816412, + -0.011002816259860992, + 0.0061003053560853004, + -0.021391529589891434, + -0.005690492689609528, + -0.002404250204563141, + 0.018496904522180557, + 0.006963520310819149, + 0.00920878816395998, + -0.019383680075407028, + 0.037107981741428375, + -0.038281023502349854, + 0.036819469183683395, + 0.038529884070158005 + ], + "wall_s": 0.10103249549865723 + }, + "epochs": [], + "evaluation_protocol": { + "test_evaluations": 0, + "test_used_for_selection": false, + "validation_evaluations": 1 + }, + "final": { + "accuracy": 0.1004, + "epoch": 0, + "evaluation_split": "validation", + "finite": true, + "loss": 16.593759375, + "step": 0 + }, + "hardware": { + "cuda_device_name": "NVIDIA GeForce GTX 1080", + "cuda_visible_devices": "5", + "device": "cuda:0", + "device_total_memory_bytes": 8507949056, + "peak_memory_allocated_bytes": 1734868480, + "peak_memory_reserved_bytes": 2690646016, + "torch_version": "2.3.1+cu118" + }, + "protocol_family": "oral_a_cifar_local_resnet_development", + "provenance": { + "git_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "git_tracked_dirty": false + }, + "schema_version": 1, + "split": { + "cifar_source_files": [ + { + "bytes": 31035704, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_1", + "sha256": "54636561a3ce25bd3e19253c6b0d8538147b0ae398331ac4a2d86c6d987368cd" + }, + { + "bytes": 31035320, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_2", + "sha256": "766b2cef9fbc745cf056b3152224f7cf77163b330ea9a15f9392beb8b89bc5a8" + }, + { + "bytes": 31035999, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_3", + "sha256": "0f00d98ebfb30b3ec0ad19f9756dc2630b89003e10525f5e148445e82aa6a1f9" + }, + { + "bytes": 31035696, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_4", + "sha256": "3f7bb240661948b8f4d53e36ec720d8306f5668bd0071dcb4e6c947f78e9682b" + }, + { + "bytes": 31035623, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_5", + "sha256": "d91802434d8376bbaeeadf58a737e3a1b12ac839077e931237e0dcd43adcb154" + }, + { + "bytes": 31035526, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/test_batch", + "sha256": "f53d8d457504f7cff4ea9e021afcf0e0ad8e24a91f3fc42091b8adef61157831" + } + ], + "dataset": "cifar10", + "evaluation_split": "validation", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "random_crop_32_padding_4_zero_then_horizontal_flip_p0.5", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + }, + "timing": { + "apical_warmup_wall_s": 6.503638505935669, + "evaluation_wall_s": 0.365323543548584, + "predictor_warmup_wall_s": 0.0, + "timing_excludes_data_loading_hashing_and_model_construction": true, + "total_timed_wall_s": 6.869091987609863, + "train_wall_s": 0.0 + }, + "work": { + "apical_macs_per_example": 202176, + "causal_scalar_observations": 200, + "components": { + "apical_projection_macs": 2565209088, + "apical_regression_macs": 2565209088, + "bp_reverse_macs_estimate": 0, + "local_weight_correlation_macs": 0, + "ordinary_forward_macs": 0, + "perturbation_forward_macs": 1029023191040, + "warmup_clean_forward_macs": 514511595520 + }, + "definition": "multiply-accumulates in conv/linear maps; one local weight correlation equals one forward-weight MAC count; BP reverse is estimated as one weight-gradient plus one activation-gradient convolution per forward convolution; elementwise nonlinearities and optimizer arithmetic excluded", + "forward_macs_per_example": 40551040, + "logical_batch_loss_queries": 200, + "per_example_cross_entropy_terms": 25376, + "total_clean_forward_examples": 12688, + "total_forward_equivalent_examples": 38064, + "total_macs_estimate": 1548665204736 + } +} diff --git a/results/oral_a_apical_screen/spatial_template_a0_etaA0.001.json b/results/oral_a_apical_screen/spatial_template_a0_etaA0.001.json new file mode 100644 index 0000000..7afb4f5 --- /dev/null +++ b/results/oral_a_apical_screen/spatial_template_a0_etaA0.001.json @@ -0,0 +1,378 @@ +{ + "apical_warmup": { + "last": { + "calibration_mse": 518.2958984375, + "prediction_target_cosine": 0.0001586056141805217, + "target_power": 518.2958984375 + }, + "steps": 100 + }, + "architecture": { + "adaptive_apical_parameters": 2260992, + "base_width": 16, + "blocks_per_stage": 3, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "depth": 20, + "family": "CIFAR 6n+2 ResNet, option-A shortcuts", + "fixed_traffic_coefficients": 188416, + "forward_parameters": 269722, + "hidden_shapes": [ + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ] + ], + "normalization": "batchnorm", + "predictor_parameters": 376832, + "residual_scale": 1.0, + "vectorizer_mode": "spatial_template", + "vectorizer_parameters": 1884160 + }, + "args": { + "a_scale": 0.0, + "a_warmup_steps": 100, + "alignment_probe": 32, + "apical_seed": null, + "augment_train": 1, + "batch_size": 128, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "data_dir": "/home/yurenh2/sdrn/data", + "depth": 20, + "device": "cuda:0", + "epochs": 0, + "eta_A": 0.001, + "eta_P": 0.01, + "eval_every": 0, + "eval_split": "validation", + "learn_P": 0, + "loader_seed": 0, + "lr": 0.03, + "lr_gamma": 0.1, + "lr_milestones": "100,150", + "lr_schedule": "constant", + "max_steps": 0, + "mode": "sdil", + "momentum": 0.9, + "normalization": "batchnorm", + "nuisance_scale": 0.0, + "out": "results/oral_a_apical_screen/spatial_template_a0_etaA0.001.json", + "output_lr": 0.1, + "pert_directions": 1, + "pert_every": 4, + "pert_sigma": 0.01, + "perturb_seed": 1000, + "predictor_warmup_steps": 0, + "residual_scale": null, + "seed": 0, + "split_seed": 2027, + "train_limit": 10000, + "use_residual": 1, + "val_examples": 5000, + "vectorizer_mode": "spatial_template", + "warmup_epochs": 0, + "weight_decay": 0.0001, + "weight_scale": 1.0, + "width": 16 + }, + "counters": { + "apical_warmup_examples": 12688, + "calibration_event_examples": 12688, + "causal_scalar_observations": 200, + "logical_batch_loss_queries": 200, + "ordinary_examples": 0, + "per_example_loss_terms": 25376, + "perturbation_events": 100, + "perturbation_forward_examples": 25376, + "predictor_warmup_examples": 0 + }, + "diagnostics": { + "early_third_mean": 8.52438776443402e-05, + "innovation_negative_gradient_cosine": [ + 0.0014789269771426916, + -0.0016939942725002766, + 0.0005816139746457338, + 0.002704082289710641, + -0.0018189277034252882, + -0.0007402379997074604, + -0.0005298232426866889, + 0.005322607234120369, + -0.0032071280293166637, + 0.0009917651768773794, + 0.008076000958681107, + 0.005940491333603859, + 0.0010138098150491714, + 0.0008690886897966266, + 0.0031331044156104326, + 0.005920985713601112, + -0.003657278837636113, + 0.002857258077710867, + 0.00569154042750597 + ], + "normalization_state": "training_batch_stats_without_running_update", + "raw_negative_gradient_cosine": [ + 0.0014789269771426916, + -0.0016939942725002766, + 0.0005816139746457338, + 0.002704082289710641, + -0.0018189277034252882, + -0.0007402379997074604, + -0.0005298232426866889, + 0.005322607234120369, + -0.0032071280293166637, + 0.0009917651768773794, + 0.008076000958681107, + 0.005940491333603859, + 0.0010138098150491714, + 0.0008690886897966266, + 0.0031331044156104326, + 0.005920985713601112, + -0.003657278837636113, + 0.002857258077710867, + 0.00569154042750597 + ], + "teaching_negative_gradient_cosine": [ + 0.0014789269771426916, + -0.0016939942725002766, + 0.0005816139746457338, + 0.002704082289710641, + -0.0018189277034252882, + -0.0007402379997074604, + -0.0005298232426866889, + 0.005322607234120369, + -0.0032071280293166637, + 0.0009917651768773794, + 0.008076000958681107, + 0.005940491333603859, + 0.0010138098150491714, + 0.0008690886897966266, + 0.0031331044156104326, + 0.005920985713601112, + -0.003657278837636113, + 0.002857258077710867, + 0.00569154042750597 + ], + "wall_s": 0.10968971252441406 + }, + "epochs": [], + "evaluation_protocol": { + "test_evaluations": 0, + "test_used_for_selection": false, + "validation_evaluations": 1 + }, + "final": { + "accuracy": 0.1004, + "epoch": 0, + "evaluation_split": "validation", + "finite": true, + "loss": 16.593759375, + "step": 0 + }, + "hardware": { + "cuda_device_name": "NVIDIA GeForce GTX 1080", + "cuda_visible_devices": "5", + "device": "cuda:0", + "device_total_memory_bytes": 8507949056, + "peak_memory_allocated_bytes": 1742341632, + "peak_memory_reserved_bytes": 2699034624, + "torch_version": "2.3.1+cu118" + }, + "protocol_family": "oral_a_cifar_local_resnet_development", + "provenance": { + "git_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "git_tracked_dirty": false + }, + "schema_version": 1, + "split": { + "cifar_source_files": [ + { + "bytes": 31035704, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_1", + "sha256": "54636561a3ce25bd3e19253c6b0d8538147b0ae398331ac4a2d86c6d987368cd" + }, + { + "bytes": 31035320, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_2", + "sha256": "766b2cef9fbc745cf056b3152224f7cf77163b330ea9a15f9392beb8b89bc5a8" + }, + { + "bytes": 31035999, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_3", + "sha256": "0f00d98ebfb30b3ec0ad19f9756dc2630b89003e10525f5e148445e82aa6a1f9" + }, + { + "bytes": 31035696, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_4", + "sha256": "3f7bb240661948b8f4d53e36ec720d8306f5668bd0071dcb4e6c947f78e9682b" + }, + { + "bytes": 31035623, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_5", + "sha256": "d91802434d8376bbaeeadf58a737e3a1b12ac839077e931237e0dcd43adcb154" + }, + { + "bytes": 31035526, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/test_batch", + "sha256": "f53d8d457504f7cff4ea9e021afcf0e0ad8e24a91f3fc42091b8adef61157831" + } + ], + "dataset": "cifar10", + "evaluation_split": "validation", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "random_crop_32_padding_4_zero_then_horizontal_flip_p0.5", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + }, + "timing": { + "apical_warmup_wall_s": 5.94947075843811, + "evaluation_wall_s": 0.36910271644592285, + "predictor_warmup_wall_s": 0.0, + "timing_excludes_data_loading_hashing_and_model_construction": true, + "total_timed_wall_s": 6.318835735321045, + "train_wall_s": 0.0 + }, + "work": { + "apical_macs_per_example": 1884160, + "causal_scalar_observations": 200, + "components": { + "apical_projection_macs": 23906222080, + "apical_regression_macs": 23906222080, + "bp_reverse_macs_estimate": 0, + "local_weight_correlation_macs": 0, + "ordinary_forward_macs": 0, + "perturbation_forward_macs": 1029023191040, + "warmup_clean_forward_macs": 514511595520 + }, + "definition": "multiply-accumulates in conv/linear maps; one local weight correlation equals one forward-weight MAC count; BP reverse is estimated as one weight-gradient plus one activation-gradient convolution per forward convolution; elementwise nonlinearities and optimizer arithmetic excluded", + "forward_macs_per_example": 40551040, + "logical_batch_loss_queries": 200, + "per_example_cross_entropy_terms": 25376, + "total_clean_forward_examples": 12688, + "total_forward_equivalent_examples": 38064, + "total_macs_estimate": 1591347230720 + } +} diff --git a/results/oral_a_apical_screen/spatial_template_a0_etaA0.01.json b/results/oral_a_apical_screen/spatial_template_a0_etaA0.01.json new file mode 100644 index 0000000..d2403d8 --- /dev/null +++ b/results/oral_a_apical_screen/spatial_template_a0_etaA0.01.json @@ -0,0 +1,378 @@ +{ + "apical_warmup": { + "last": { + "calibration_mse": 518.2982058317765, + "prediction_target_cosine": 0.00015677468540085558, + "target_power": 518.2958984375 + }, + "steps": 100 + }, + "architecture": { + "adaptive_apical_parameters": 2260992, + "base_width": 16, + "blocks_per_stage": 3, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "depth": 20, + "family": "CIFAR 6n+2 ResNet, option-A shortcuts", + "fixed_traffic_coefficients": 188416, + "forward_parameters": 269722, + "hidden_shapes": [ + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ] + ], + "normalization": "batchnorm", + "predictor_parameters": 376832, + "residual_scale": 1.0, + "vectorizer_mode": "spatial_template", + "vectorizer_parameters": 1884160 + }, + "args": { + "a_scale": 0.0, + "a_warmup_steps": 100, + "alignment_probe": 32, + "apical_seed": null, + "augment_train": 1, + "batch_size": 128, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "data_dir": "/home/yurenh2/sdrn/data", + "depth": 20, + "device": "cuda:0", + "epochs": 0, + "eta_A": 0.01, + "eta_P": 0.01, + "eval_every": 0, + "eval_split": "validation", + "learn_P": 0, + "loader_seed": 0, + "lr": 0.03, + "lr_gamma": 0.1, + "lr_milestones": "100,150", + "lr_schedule": "constant", + "max_steps": 0, + "mode": "sdil", + "momentum": 0.9, + "normalization": "batchnorm", + "nuisance_scale": 0.0, + "out": "results/oral_a_apical_screen/spatial_template_a0_etaA0.01.json", + "output_lr": 0.1, + "pert_directions": 1, + "pert_every": 4, + "pert_sigma": 0.01, + "perturb_seed": 1000, + "predictor_warmup_steps": 0, + "residual_scale": null, + "seed": 0, + "split_seed": 2027, + "train_limit": 10000, + "use_residual": 1, + "val_examples": 5000, + "vectorizer_mode": "spatial_template", + "warmup_epochs": 0, + "weight_decay": 0.0001, + "weight_scale": 1.0, + "width": 16 + }, + "counters": { + "apical_warmup_examples": 12688, + "calibration_event_examples": 12688, + "causal_scalar_observations": 200, + "logical_batch_loss_queries": 200, + "ordinary_examples": 0, + "per_example_loss_terms": 25376, + "perturbation_events": 100, + "perturbation_forward_examples": 25376, + "predictor_warmup_examples": 0 + }, + "diagnostics": { + "early_third_mean": 5.210153176449239e-05, + "innovation_negative_gradient_cosine": [ + 0.001460368512198329, + -0.001773762283846736, + 0.0005576296243816614, + 0.002642326755449176, + -0.0018755876226350665, + -0.0006983657949604094, + -0.0005098751280456781, + 0.0053292931988835335, + -0.003289436688646674, + 0.0009565821383148432, + 0.007996374741196632, + 0.005934947170317173, + 0.0008049617754295468, + 0.0010773323010653257, + 0.0031135338358581066, + 0.00604074215516448, + -0.0038750299718230963, + 0.0030057388357818127, + 0.005638218019157648 + ], + "normalization_state": "training_batch_stats_without_running_update", + "raw_negative_gradient_cosine": [ + 0.001460368512198329, + -0.001773762283846736, + 0.0005576296243816614, + 0.002642326755449176, + -0.0018755876226350665, + -0.0006983657949604094, + -0.0005098751280456781, + 0.0053292931988835335, + -0.003289436688646674, + 0.0009565821383148432, + 0.007996374741196632, + 0.005934947170317173, + 0.0008049617754295468, + 0.0010773323010653257, + 0.0031135338358581066, + 0.00604074215516448, + -0.0038750299718230963, + 0.0030057388357818127, + 0.005638218019157648 + ], + "teaching_negative_gradient_cosine": [ + 0.001460368512198329, + -0.001773762283846736, + 0.0005576296243816614, + 0.002642326755449176, + -0.0018755876226350665, + -0.0006983657949604094, + -0.0005098751280456781, + 0.0053292931988835335, + -0.003289436688646674, + 0.0009565821383148432, + 0.007996374741196632, + 0.005934947170317173, + 0.0008049617754295468, + 0.0010773323010653257, + 0.0031135338358581066, + 0.00604074215516448, + -0.0038750299718230963, + 0.0030057388357818127, + 0.005638218019157648 + ], + "wall_s": 0.09761571884155273 + }, + "epochs": [], + "evaluation_protocol": { + "test_evaluations": 0, + "test_used_for_selection": false, + "validation_evaluations": 1 + }, + "final": { + "accuracy": 0.1004, + "epoch": 0, + "evaluation_split": "validation", + "finite": true, + "loss": 16.593759375, + "step": 0 + }, + "hardware": { + "cuda_device_name": "NVIDIA GeForce GTX 1080", + "cuda_visible_devices": "5", + "device": "cuda:0", + "device_total_memory_bytes": 8507949056, + "peak_memory_allocated_bytes": 1742341632, + "peak_memory_reserved_bytes": 2699034624, + "torch_version": "2.3.1+cu118" + }, + "protocol_family": "oral_a_cifar_local_resnet_development", + "provenance": { + "git_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "git_tracked_dirty": false + }, + "schema_version": 1, + "split": { + "cifar_source_files": [ + { + "bytes": 31035704, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_1", + "sha256": "54636561a3ce25bd3e19253c6b0d8538147b0ae398331ac4a2d86c6d987368cd" + }, + { + "bytes": 31035320, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_2", + "sha256": "766b2cef9fbc745cf056b3152224f7cf77163b330ea9a15f9392beb8b89bc5a8" + }, + { + "bytes": 31035999, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_3", + "sha256": "0f00d98ebfb30b3ec0ad19f9756dc2630b89003e10525f5e148445e82aa6a1f9" + }, + { + "bytes": 31035696, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_4", + "sha256": "3f7bb240661948b8f4d53e36ec720d8306f5668bd0071dcb4e6c947f78e9682b" + }, + { + "bytes": 31035623, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_5", + "sha256": "d91802434d8376bbaeeadf58a737e3a1b12ac839077e931237e0dcd43adcb154" + }, + { + "bytes": 31035526, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/test_batch", + "sha256": "f53d8d457504f7cff4ea9e021afcf0e0ad8e24a91f3fc42091b8adef61157831" + } + ], + "dataset": "cifar10", + "evaluation_split": "validation", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "random_crop_32_padding_4_zero_then_horizontal_flip_p0.5", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + }, + "timing": { + "apical_warmup_wall_s": 5.951646089553833, + "evaluation_wall_s": 0.37953901290893555, + "predictor_warmup_wall_s": 0.0, + "timing_excludes_data_loading_hashing_and_model_construction": true, + "total_timed_wall_s": 6.331512212753296, + "train_wall_s": 0.0 + }, + "work": { + "apical_macs_per_example": 1884160, + "causal_scalar_observations": 200, + "components": { + "apical_projection_macs": 23906222080, + "apical_regression_macs": 23906222080, + "bp_reverse_macs_estimate": 0, + "local_weight_correlation_macs": 0, + "ordinary_forward_macs": 0, + "perturbation_forward_macs": 1029023191040, + "warmup_clean_forward_macs": 514511595520 + }, + "definition": "multiply-accumulates in conv/linear maps; one local weight correlation equals one forward-weight MAC count; BP reverse is estimated as one weight-gradient plus one activation-gradient convolution per forward convolution; elementwise nonlinearities and optimizer arithmetic excluded", + "forward_macs_per_example": 40551040, + "logical_batch_loss_queries": 200, + "per_example_cross_entropy_terms": 25376, + "total_clean_forward_examples": 12688, + "total_forward_equivalent_examples": 38064, + "total_macs_estimate": 1591347230720 + } +} diff --git a/results/oral_a_apical_screen/spatial_template_a0_etaA0.1.json b/results/oral_a_apical_screen/spatial_template_a0_etaA0.1.json new file mode 100644 index 0000000..aa92166 --- /dev/null +++ b/results/oral_a_apical_screen/spatial_template_a0_etaA0.1.json @@ -0,0 +1,378 @@ +{ + "apical_warmup": { + "last": { + "calibration_mse": 518.4102902619735, + "prediction_target_cosine": 0.00016023495648166922, + "target_power": 518.2958984375 + }, + "steps": 100 + }, + "architecture": { + "adaptive_apical_parameters": 2260992, + "base_width": 16, + "blocks_per_stage": 3, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "depth": 20, + "family": "CIFAR 6n+2 ResNet, option-A shortcuts", + "fixed_traffic_coefficients": 188416, + "forward_parameters": 269722, + "hidden_shapes": [ + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ] + ], + "normalization": "batchnorm", + "predictor_parameters": 376832, + "residual_scale": 1.0, + "vectorizer_mode": "spatial_template", + "vectorizer_parameters": 1884160 + }, + "args": { + "a_scale": 0.0, + "a_warmup_steps": 100, + "alignment_probe": 32, + "apical_seed": null, + "augment_train": 1, + "batch_size": 128, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "data_dir": "/home/yurenh2/sdrn/data", + "depth": 20, + "device": "cuda:0", + "epochs": 0, + "eta_A": 0.1, + "eta_P": 0.01, + "eval_every": 0, + "eval_split": "validation", + "learn_P": 0, + "loader_seed": 0, + "lr": 0.03, + "lr_gamma": 0.1, + "lr_milestones": "100,150", + "lr_schedule": "constant", + "max_steps": 0, + "mode": "sdil", + "momentum": 0.9, + "normalization": "batchnorm", + "nuisance_scale": 0.0, + "out": "results/oral_a_apical_screen/spatial_template_a0_etaA0.1.json", + "output_lr": 0.1, + "pert_directions": 1, + "pert_every": 4, + "pert_sigma": 0.01, + "perturb_seed": 1000, + "predictor_warmup_steps": 0, + "residual_scale": null, + "seed": 0, + "split_seed": 2027, + "train_limit": 10000, + "use_residual": 1, + "val_examples": 5000, + "vectorizer_mode": "spatial_template", + "warmup_epochs": 0, + "weight_decay": 0.0001, + "weight_scale": 1.0, + "width": 16 + }, + "counters": { + "apical_warmup_examples": 12688, + "calibration_event_examples": 12688, + "causal_scalar_observations": 200, + "logical_batch_loss_queries": 200, + "ordinary_examples": 0, + "per_example_loss_terms": 25376, + "perturbation_events": 100, + "perturbation_forward_examples": 25376, + "predictor_warmup_examples": 0 + }, + "diagnostics": { + "early_third_mean": -0.00022858582572856298, + "innovation_negative_gradient_cosine": [ + 0.0013673536013811827, + -0.0023725673090666533, + 0.0003463272878434509, + 0.002020572777837515, + -0.002358220284804702, + -0.0003749810275621712, + -0.00031586794648319483, + 0.005033853929489851, + -0.003750653937458992, + 0.0008844805415719748, + 0.0069566755555570126, + 0.0055755311623215675, + -0.0006150356493890285, + 0.0025809917133301497, + 0.002333217067644, + 0.006841001100838184, + -0.0056769344955682755, + 0.0036507542245090008, + 0.005224898923188448 + ], + "normalization_state": "training_batch_stats_without_running_update", + "raw_negative_gradient_cosine": [ + 0.0013673536013811827, + -0.0023725673090666533, + 0.0003463272878434509, + 0.002020572777837515, + -0.002358220284804702, + -0.0003749810275621712, + -0.00031586794648319483, + 0.005033853929489851, + -0.003750653937458992, + 0.0008844805415719748, + 0.0069566755555570126, + 0.0055755311623215675, + -0.0006150356493890285, + 0.0025809917133301497, + 0.002333217067644, + 0.006841001100838184, + -0.0056769344955682755, + 0.0036507542245090008, + 0.005224898923188448 + ], + "teaching_negative_gradient_cosine": [ + 0.0013673536013811827, + -0.0023725673090666533, + 0.0003463272878434509, + 0.002020572777837515, + -0.002358220284804702, + -0.0003749810275621712, + -0.00031586794648319483, + 0.005033853929489851, + -0.003750653937458992, + 0.0008844805415719748, + 0.0069566755555570126, + 0.0055755311623215675, + -0.0006150356493890285, + 0.0025809917133301497, + 0.002333217067644, + 0.006841001100838184, + -0.0056769344955682755, + 0.0036507542245090008, + 0.005224898923188448 + ], + "wall_s": 0.1099550724029541 + }, + "epochs": [], + "evaluation_protocol": { + "test_evaluations": 0, + "test_used_for_selection": false, + "validation_evaluations": 1 + }, + "final": { + "accuracy": 0.1004, + "epoch": 0, + "evaluation_split": "validation", + "finite": true, + "loss": 16.593759375, + "step": 0 + }, + "hardware": { + "cuda_device_name": "NVIDIA GeForce GTX 1080", + "cuda_visible_devices": "5", + "device": "cuda:0", + "device_total_memory_bytes": 8507949056, + "peak_memory_allocated_bytes": 1742341632, + "peak_memory_reserved_bytes": 2699034624, + "torch_version": "2.3.1+cu118" + }, + "protocol_family": "oral_a_cifar_local_resnet_development", + "provenance": { + "git_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "git_tracked_dirty": false + }, + "schema_version": 1, + "split": { + "cifar_source_files": [ + { + "bytes": 31035704, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_1", + "sha256": "54636561a3ce25bd3e19253c6b0d8538147b0ae398331ac4a2d86c6d987368cd" + }, + { + "bytes": 31035320, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_2", + "sha256": "766b2cef9fbc745cf056b3152224f7cf77163b330ea9a15f9392beb8b89bc5a8" + }, + { + "bytes": 31035999, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_3", + "sha256": "0f00d98ebfb30b3ec0ad19f9756dc2630b89003e10525f5e148445e82aa6a1f9" + }, + { + "bytes": 31035696, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_4", + "sha256": "3f7bb240661948b8f4d53e36ec720d8306f5668bd0071dcb4e6c947f78e9682b" + }, + { + "bytes": 31035623, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_5", + "sha256": "d91802434d8376bbaeeadf58a737e3a1b12ac839077e931237e0dcd43adcb154" + }, + { + "bytes": 31035526, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/test_batch", + "sha256": "f53d8d457504f7cff4ea9e021afcf0e0ad8e24a91f3fc42091b8adef61157831" + } + ], + "dataset": "cifar10", + "evaluation_split": "validation", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "random_crop_32_padding_4_zero_then_horizontal_flip_p0.5", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + }, + "timing": { + "apical_warmup_wall_s": 5.992250204086304, + "evaluation_wall_s": 0.3817732334136963, + "predictor_warmup_wall_s": 0.0, + "timing_excludes_data_loading_hashing_and_model_construction": true, + "total_timed_wall_s": 6.3742194175720215, + "train_wall_s": 0.0 + }, + "work": { + "apical_macs_per_example": 1884160, + "causal_scalar_observations": 200, + "components": { + "apical_projection_macs": 23906222080, + "apical_regression_macs": 23906222080, + "bp_reverse_macs_estimate": 0, + "local_weight_correlation_macs": 0, + "ordinary_forward_macs": 0, + "perturbation_forward_macs": 1029023191040, + "warmup_clean_forward_macs": 514511595520 + }, + "definition": "multiply-accumulates in conv/linear maps; one local weight correlation equals one forward-weight MAC count; BP reverse is estimated as one weight-gradient plus one activation-gradient convolution per forward convolution; elementwise nonlinearities and optimizer arithmetic excluded", + "forward_macs_per_example": 40551040, + "logical_batch_loss_queries": 200, + "per_example_cross_entropy_terms": 25376, + "total_clean_forward_examples": 12688, + "total_forward_equivalent_examples": 38064, + "total_macs_estimate": 1591347230720 + } +} diff --git a/results/oral_a_apical_screen/spatial_template_a1_etaA0.001.json b/results/oral_a_apical_screen/spatial_template_a1_etaA0.001.json new file mode 100644 index 0000000..fbb1a3e --- /dev/null +++ b/results/oral_a_apical_screen/spatial_template_a1_etaA0.001.json @@ -0,0 +1,378 @@ +{ + "apical_warmup": { + "last": { + "calibration_mse": 518.2959063986074, + "prediction_target_cosine": 0.00011094105930582137, + "target_power": 518.2958984375 + }, + "steps": 100 + }, + "architecture": { + "adaptive_apical_parameters": 2260992, + "base_width": 16, + "blocks_per_stage": 3, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "depth": 20, + "family": "CIFAR 6n+2 ResNet, option-A shortcuts", + "fixed_traffic_coefficients": 188416, + "forward_parameters": 269722, + "hidden_shapes": [ + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ] + ], + "normalization": "batchnorm", + "predictor_parameters": 376832, + "residual_scale": 1.0, + "vectorizer_mode": "spatial_template", + "vectorizer_parameters": 1884160 + }, + "args": { + "a_scale": 1.0, + "a_warmup_steps": 100, + "alignment_probe": 32, + "apical_seed": null, + "augment_train": 1, + "batch_size": 128, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "data_dir": "/home/yurenh2/sdrn/data", + "depth": 20, + "device": "cuda:0", + "epochs": 0, + "eta_A": 0.001, + "eta_P": 0.01, + "eval_every": 0, + "eval_split": "validation", + "learn_P": 0, + "loader_seed": 0, + "lr": 0.03, + "lr_gamma": 0.1, + "lr_milestones": "100,150", + "lr_schedule": "constant", + "max_steps": 0, + "mode": "sdil", + "momentum": 0.9, + "normalization": "batchnorm", + "nuisance_scale": 0.0, + "out": "results/oral_a_apical_screen/spatial_template_a1_etaA0.001.json", + "output_lr": 0.1, + "pert_directions": 1, + "pert_every": 4, + "pert_sigma": 0.01, + "perturb_seed": 1000, + "predictor_warmup_steps": 0, + "residual_scale": null, + "seed": 0, + "split_seed": 2027, + "train_limit": 10000, + "use_residual": 1, + "val_examples": 5000, + "vectorizer_mode": "spatial_template", + "warmup_epochs": 0, + "weight_decay": 0.0001, + "weight_scale": 1.0, + "width": 16 + }, + "counters": { + "apical_warmup_examples": 12688, + "calibration_event_examples": 12688, + "causal_scalar_observations": 200, + "logical_batch_loss_queries": 200, + "ordinary_examples": 0, + "per_example_loss_terms": 25376, + "perturbation_events": 100, + "perturbation_forward_examples": 25376, + "predictor_warmup_examples": 0 + }, + "diagnostics": { + "early_third_mean": 2.9081663039202493e-05, + "innovation_negative_gradient_cosine": [ + 0.0015069551300257444, + -0.0017042217077687383, + 0.000544557988177985, + 0.002581007080152631, + -0.0019102504011243582, + -0.0008435581112280488, + -0.000518693879712373, + 0.004991275258362293, + -0.0025019566528499126, + 0.001231340691447258, + 0.007521952502429485, + 0.006807704456150532, + 0.0011750941630452871, + 0.006320856511592865, + -0.001078901463188231, + 0.003498366568237543, + -0.0007391999242827296, + 0.0036910739727318287, + 0.0013871951960027218 + ], + "normalization_state": "training_batch_stats_without_running_update", + "raw_negative_gradient_cosine": [ + 0.0015069551300257444, + -0.0017042217077687383, + 0.000544557988177985, + 0.002581007080152631, + -0.0019102504011243582, + -0.0008435581112280488, + -0.000518693879712373, + 0.004991275258362293, + -0.0025019566528499126, + 0.001231340691447258, + 0.007521952502429485, + 0.006807704456150532, + 0.0011750941630452871, + 0.006320856511592865, + -0.001078901463188231, + 0.003498366568237543, + -0.0007391999242827296, + 0.0036910739727318287, + 0.0013871951960027218 + ], + "teaching_negative_gradient_cosine": [ + 0.0015069551300257444, + -0.0017042217077687383, + 0.000544557988177985, + 0.002581007080152631, + -0.0019102504011243582, + -0.0008435581112280488, + -0.000518693879712373, + 0.004991275258362293, + -0.0025019566528499126, + 0.001231340691447258, + 0.007521952502429485, + 0.006807704456150532, + 0.0011750941630452871, + 0.006320856511592865, + -0.001078901463188231, + 0.003498366568237543, + -0.0007391999242827296, + 0.0036910739727318287, + 0.0013871951960027218 + ], + "wall_s": 0.10784101486206055 + }, + "epochs": [], + "evaluation_protocol": { + "test_evaluations": 0, + "test_used_for_selection": false, + "validation_evaluations": 1 + }, + "final": { + "accuracy": 0.1004, + "epoch": 0, + "evaluation_split": "validation", + "finite": true, + "loss": 16.593759375, + "step": 0 + }, + "hardware": { + "cuda_device_name": "NVIDIA GeForce GTX 1080", + "cuda_visible_devices": "5", + "device": "cuda:0", + "device_total_memory_bytes": 8507949056, + "peak_memory_allocated_bytes": 1742341632, + "peak_memory_reserved_bytes": 2699034624, + "torch_version": "2.3.1+cu118" + }, + "protocol_family": "oral_a_cifar_local_resnet_development", + "provenance": { + "git_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "git_tracked_dirty": false + }, + "schema_version": 1, + "split": { + "cifar_source_files": [ + { + "bytes": 31035704, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_1", + "sha256": "54636561a3ce25bd3e19253c6b0d8538147b0ae398331ac4a2d86c6d987368cd" + }, + { + "bytes": 31035320, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_2", + "sha256": "766b2cef9fbc745cf056b3152224f7cf77163b330ea9a15f9392beb8b89bc5a8" + }, + { + "bytes": 31035999, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_3", + "sha256": "0f00d98ebfb30b3ec0ad19f9756dc2630b89003e10525f5e148445e82aa6a1f9" + }, + { + "bytes": 31035696, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_4", + "sha256": "3f7bb240661948b8f4d53e36ec720d8306f5668bd0071dcb4e6c947f78e9682b" + }, + { + "bytes": 31035623, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_5", + "sha256": "d91802434d8376bbaeeadf58a737e3a1b12ac839077e931237e0dcd43adcb154" + }, + { + "bytes": 31035526, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/test_batch", + "sha256": "f53d8d457504f7cff4ea9e021afcf0e0ad8e24a91f3fc42091b8adef61157831" + } + ], + "dataset": "cifar10", + "evaluation_split": "validation", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "random_crop_32_padding_4_zero_then_horizontal_flip_p0.5", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + }, + "timing": { + "apical_warmup_wall_s": 5.946948766708374, + "evaluation_wall_s": 0.3671278953552246, + "predictor_warmup_wall_s": 0.0, + "timing_excludes_data_loading_hashing_and_model_construction": true, + "total_timed_wall_s": 6.3143370151519775, + "train_wall_s": 0.0 + }, + "work": { + "apical_macs_per_example": 1884160, + "causal_scalar_observations": 200, + "components": { + "apical_projection_macs": 23906222080, + "apical_regression_macs": 23906222080, + "bp_reverse_macs_estimate": 0, + "local_weight_correlation_macs": 0, + "ordinary_forward_macs": 0, + "perturbation_forward_macs": 1029023191040, + "warmup_clean_forward_macs": 514511595520 + }, + "definition": "multiply-accumulates in conv/linear maps; one local weight correlation equals one forward-weight MAC count; BP reverse is estimated as one weight-gradient plus one activation-gradient convolution per forward convolution; elementwise nonlinearities and optimizer arithmetic excluded", + "forward_macs_per_example": 40551040, + "logical_batch_loss_queries": 200, + "per_example_cross_entropy_terms": 25376, + "total_clean_forward_examples": 12688, + "total_forward_equivalent_examples": 38064, + "total_macs_estimate": 1591347230720 + } +} diff --git a/results/oral_a_apical_screen/spatial_template_a1_etaA0.01.json b/results/oral_a_apical_screen/spatial_template_a1_etaA0.01.json new file mode 100644 index 0000000..37b08b9 --- /dev/null +++ b/results/oral_a_apical_screen/spatial_template_a1_etaA0.01.json @@ -0,0 +1,378 @@ +{ + "apical_warmup": { + "last": { + "calibration_mse": 518.2982217539911, + "prediction_target_cosine": 0.0001529906778155134, + "target_power": 518.2958984375 + }, + "steps": 100 + }, + "architecture": { + "adaptive_apical_parameters": 2260992, + "base_width": 16, + "blocks_per_stage": 3, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "depth": 20, + "family": "CIFAR 6n+2 ResNet, option-A shortcuts", + "fixed_traffic_coefficients": 188416, + "forward_parameters": 269722, + "hidden_shapes": [ + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ] + ], + "normalization": "batchnorm", + "predictor_parameters": 376832, + "residual_scale": 1.0, + "vectorizer_mode": "spatial_template", + "vectorizer_parameters": 1884160 + }, + "args": { + "a_scale": 1.0, + "a_warmup_steps": 100, + "alignment_probe": 32, + "apical_seed": null, + "augment_train": 1, + "batch_size": 128, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "data_dir": "/home/yurenh2/sdrn/data", + "depth": 20, + "device": "cuda:0", + "epochs": 0, + "eta_A": 0.01, + "eta_P": 0.01, + "eval_every": 0, + "eval_split": "validation", + "learn_P": 0, + "loader_seed": 0, + "lr": 0.03, + "lr_gamma": 0.1, + "lr_milestones": "100,150", + "lr_schedule": "constant", + "max_steps": 0, + "mode": "sdil", + "momentum": 0.9, + "normalization": "batchnorm", + "nuisance_scale": 0.0, + "out": "results/oral_a_apical_screen/spatial_template_a1_etaA0.01.json", + "output_lr": 0.1, + "pert_directions": 1, + "pert_every": 4, + "pert_sigma": 0.01, + "perturb_seed": 1000, + "predictor_warmup_steps": 0, + "residual_scale": null, + "seed": 0, + "split_seed": 2027, + "train_limit": 10000, + "use_residual": 1, + "val_examples": 5000, + "vectorizer_mode": "spatial_template", + "warmup_epochs": 0, + "weight_decay": 0.0001, + "weight_scale": 1.0, + "width": 16 + }, + "counters": { + "apical_warmup_examples": 12688, + "calibration_event_examples": 12688, + "causal_scalar_observations": 200, + "logical_batch_loss_queries": 200, + "ordinary_examples": 0, + "per_example_loss_terms": 25376, + "perturbation_events": 100, + "perturbation_forward_examples": 25376, + "predictor_warmup_examples": 0 + }, + "diagnostics": { + "early_third_mean": 4.663398916212221e-05, + "innovation_negative_gradient_cosine": [ + 0.0014630088116973639, + -0.0017752200365066528, + 0.0005540486890822649, + 0.0026310232933610678, + -0.001884695258922875, + -0.0007083615637384355, + -0.0005085690645501018, + 0.0053089335560798645, + -0.0032270257361233234, + 0.0009825218003243208, + 0.007963800802826881, + 0.0060312747955322266, + 0.0008216925198212266, + 0.0017954821232706308, + 0.0026481396052986383, + 0.0059055485762655735, + -0.0036025659646838903, + 0.0031842084135860205, + 0.005277156364172697 + ], + "normalization_state": "training_batch_stats_without_running_update", + "raw_negative_gradient_cosine": [ + 0.0014630088116973639, + -0.0017752200365066528, + 0.0005540486890822649, + 0.0026310232933610678, + -0.001884695258922875, + -0.0007083615637384355, + -0.0005085690645501018, + 0.0053089335560798645, + -0.0032270257361233234, + 0.0009825218003243208, + 0.007963800802826881, + 0.0060312747955322266, + 0.0008216925198212266, + 0.0017954821232706308, + 0.0026481396052986383, + 0.0059055485762655735, + -0.0036025659646838903, + 0.0031842084135860205, + 0.005277156364172697 + ], + "teaching_negative_gradient_cosine": [ + 0.0014630088116973639, + -0.0017752200365066528, + 0.0005540486890822649, + 0.0026310232933610678, + -0.001884695258922875, + -0.0007083615637384355, + -0.0005085690645501018, + 0.0053089335560798645, + -0.0032270257361233234, + 0.0009825218003243208, + 0.007963800802826881, + 0.0060312747955322266, + 0.0008216925198212266, + 0.0017954821232706308, + 0.0026481396052986383, + 0.0059055485762655735, + -0.0036025659646838903, + 0.0031842084135860205, + 0.005277156364172697 + ], + "wall_s": 0.11437058448791504 + }, + "epochs": [], + "evaluation_protocol": { + "test_evaluations": 0, + "test_used_for_selection": false, + "validation_evaluations": 1 + }, + "final": { + "accuracy": 0.1004, + "epoch": 0, + "evaluation_split": "validation", + "finite": true, + "loss": 16.593759375, + "step": 0 + }, + "hardware": { + "cuda_device_name": "NVIDIA GeForce GTX 1080", + "cuda_visible_devices": "5", + "device": "cuda:0", + "device_total_memory_bytes": 8507949056, + "peak_memory_allocated_bytes": 1742341632, + "peak_memory_reserved_bytes": 2699034624, + "torch_version": "2.3.1+cu118" + }, + "protocol_family": "oral_a_cifar_local_resnet_development", + "provenance": { + "git_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "git_tracked_dirty": false + }, + "schema_version": 1, + "split": { + "cifar_source_files": [ + { + "bytes": 31035704, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_1", + "sha256": "54636561a3ce25bd3e19253c6b0d8538147b0ae398331ac4a2d86c6d987368cd" + }, + { + "bytes": 31035320, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_2", + "sha256": "766b2cef9fbc745cf056b3152224f7cf77163b330ea9a15f9392beb8b89bc5a8" + }, + { + "bytes": 31035999, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_3", + "sha256": "0f00d98ebfb30b3ec0ad19f9756dc2630b89003e10525f5e148445e82aa6a1f9" + }, + { + "bytes": 31035696, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_4", + "sha256": "3f7bb240661948b8f4d53e36ec720d8306f5668bd0071dcb4e6c947f78e9682b" + }, + { + "bytes": 31035623, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_5", + "sha256": "d91802434d8376bbaeeadf58a737e3a1b12ac839077e931237e0dcd43adcb154" + }, + { + "bytes": 31035526, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/test_batch", + "sha256": "f53d8d457504f7cff4ea9e021afcf0e0ad8e24a91f3fc42091b8adef61157831" + } + ], + "dataset": "cifar10", + "evaluation_split": "validation", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "random_crop_32_padding_4_zero_then_horizontal_flip_p0.5", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + }, + "timing": { + "apical_warmup_wall_s": 5.952100038528442, + "evaluation_wall_s": 0.3667166233062744, + "predictor_warmup_wall_s": 0.0, + "timing_excludes_data_loading_hashing_and_model_construction": true, + "total_timed_wall_s": 6.318997144699097, + "train_wall_s": 0.0 + }, + "work": { + "apical_macs_per_example": 1884160, + "causal_scalar_observations": 200, + "components": { + "apical_projection_macs": 23906222080, + "apical_regression_macs": 23906222080, + "bp_reverse_macs_estimate": 0, + "local_weight_correlation_macs": 0, + "ordinary_forward_macs": 0, + "perturbation_forward_macs": 1029023191040, + "warmup_clean_forward_macs": 514511595520 + }, + "definition": "multiply-accumulates in conv/linear maps; one local weight correlation equals one forward-weight MAC count; BP reverse is estimated as one weight-gradient plus one activation-gradient convolution per forward convolution; elementwise nonlinearities and optimizer arithmetic excluded", + "forward_macs_per_example": 40551040, + "logical_batch_loss_queries": 200, + "per_example_cross_entropy_terms": 25376, + "total_clean_forward_examples": 12688, + "total_forward_equivalent_examples": 38064, + "total_macs_estimate": 1591347230720 + } +} diff --git a/results/oral_a_apical_screen/spatial_template_a1_etaA0.1.json b/results/oral_a_apical_screen/spatial_template_a1_etaA0.1.json new file mode 100644 index 0000000..4ae151b --- /dev/null +++ b/results/oral_a_apical_screen/spatial_template_a1_etaA0.1.json @@ -0,0 +1,378 @@ +{ + "apical_warmup": { + "last": { + "calibration_mse": 518.4102889351223, + "prediction_target_cosine": 0.00016010901837600736, + "target_power": 518.2958984375 + }, + "steps": 100 + }, + "architecture": { + "adaptive_apical_parameters": 2260992, + "base_width": 16, + "blocks_per_stage": 3, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "depth": 20, + "family": "CIFAR 6n+2 ResNet, option-A shortcuts", + "fixed_traffic_coefficients": 188416, + "forward_parameters": 269722, + "hidden_shapes": [ + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 16, + 32, + 32 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 32, + 16, + 16 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ], + [ + 64, + 8, + 8 + ] + ], + "normalization": "batchnorm", + "predictor_parameters": 376832, + "residual_scale": 1.0, + "vectorizer_mode": "spatial_template", + "vectorizer_parameters": 1884160 + }, + "args": { + "a_scale": 1.0, + "a_warmup_steps": 100, + "alignment_probe": 32, + "apical_seed": null, + "augment_train": 1, + "batch_size": 128, + "bn_eps": 1e-05, + "bn_momentum": 0.1, + "data_dir": "/home/yurenh2/sdrn/data", + "depth": 20, + "device": "cuda:0", + "epochs": 0, + "eta_A": 0.1, + "eta_P": 0.01, + "eval_every": 0, + "eval_split": "validation", + "learn_P": 0, + "loader_seed": 0, + "lr": 0.03, + "lr_gamma": 0.1, + "lr_milestones": "100,150", + "lr_schedule": "constant", + "max_steps": 0, + "mode": "sdil", + "momentum": 0.9, + "normalization": "batchnorm", + "nuisance_scale": 0.0, + "out": "results/oral_a_apical_screen/spatial_template_a1_etaA0.1.json", + "output_lr": 0.1, + "pert_directions": 1, + "pert_every": 4, + "pert_sigma": 0.01, + "perturb_seed": 1000, + "predictor_warmup_steps": 0, + "residual_scale": null, + "seed": 0, + "split_seed": 2027, + "train_limit": 10000, + "use_residual": 1, + "val_examples": 5000, + "vectorizer_mode": "spatial_template", + "warmup_epochs": 0, + "weight_decay": 0.0001, + "weight_scale": 1.0, + "width": 16 + }, + "counters": { + "apical_warmup_examples": 12688, + "calibration_event_examples": 12688, + "causal_scalar_observations": 200, + "logical_batch_loss_queries": 200, + "ordinary_examples": 0, + "per_example_loss_terms": 25376, + "perturbation_events": 100, + "perturbation_forward_examples": 25376, + "predictor_warmup_examples": 0 + }, + "diagnostics": { + "early_third_mean": -0.00022893324785400182, + "innovation_negative_gradient_cosine": [ + 0.0013674263609573245, + -0.002372740302234888, + 0.00034607702400535345, + 0.002019967418164015, + -0.002358763013035059, + -0.00037556697498075664, + -0.0003156884922645986, + 0.005032744724303484, + -0.003746849950402975, + 0.0008861758979037404, + 0.0069563621655106544, + 0.0055808257311582565, + -0.0006146630621515214, + 0.002619678620249033, + 0.002311413176357746, + 0.006832084618508816, + -0.005659564398229122, + 0.0036577084101736546, + 0.005209040828049183 + ], + "normalization_state": "training_batch_stats_without_running_update", + "raw_negative_gradient_cosine": [ + 0.0013674263609573245, + -0.002372740302234888, + 0.00034607702400535345, + 0.002019967418164015, + -0.002358763013035059, + -0.00037556697498075664, + -0.0003156884922645986, + 0.005032744724303484, + -0.003746849950402975, + 0.0008861758979037404, + 0.0069563621655106544, + 0.0055808257311582565, + -0.0006146630621515214, + 0.002619678620249033, + 0.002311413176357746, + 0.006832084618508816, + -0.005659564398229122, + 0.0036577084101736546, + 0.005209040828049183 + ], + "teaching_negative_gradient_cosine": [ + 0.0013674263609573245, + -0.002372740302234888, + 0.00034607702400535345, + 0.002019967418164015, + -0.002358763013035059, + -0.00037556697498075664, + -0.0003156884922645986, + 0.005032744724303484, + -0.003746849950402975, + 0.0008861758979037404, + 0.0069563621655106544, + 0.0055808257311582565, + -0.0006146630621515214, + 0.002619678620249033, + 0.002311413176357746, + 0.006832084618508816, + -0.005659564398229122, + 0.0036577084101736546, + 0.005209040828049183 + ], + "wall_s": 0.10603523254394531 + }, + "epochs": [], + "evaluation_protocol": { + "test_evaluations": 0, + "test_used_for_selection": false, + "validation_evaluations": 1 + }, + "final": { + "accuracy": 0.1004, + "epoch": 0, + "evaluation_split": "validation", + "finite": true, + "loss": 16.593759375, + "step": 0 + }, + "hardware": { + "cuda_device_name": "NVIDIA GeForce GTX 1080", + "cuda_visible_devices": "5", + "device": "cuda:0", + "device_total_memory_bytes": 8507949056, + "peak_memory_allocated_bytes": 1742341632, + "peak_memory_reserved_bytes": 2699034624, + "torch_version": "2.3.1+cu118" + }, + "protocol_family": "oral_a_cifar_local_resnet_development", + "provenance": { + "git_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "git_tracked_dirty": false + }, + "schema_version": 1, + "split": { + "cifar_source_files": [ + { + "bytes": 31035704, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_1", + "sha256": "54636561a3ce25bd3e19253c6b0d8538147b0ae398331ac4a2d86c6d987368cd" + }, + { + "bytes": 31035320, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_2", + "sha256": "766b2cef9fbc745cf056b3152224f7cf77163b330ea9a15f9392beb8b89bc5a8" + }, + { + "bytes": 31035999, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_3", + "sha256": "0f00d98ebfb30b3ec0ad19f9756dc2630b89003e10525f5e148445e82aa6a1f9" + }, + { + "bytes": 31035696, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_4", + "sha256": "3f7bb240661948b8f4d53e36ec720d8306f5668bd0071dcb4e6c947f78e9682b" + }, + { + "bytes": 31035623, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_5", + "sha256": "d91802434d8376bbaeeadf58a737e3a1b12ac839077e931237e0dcd43adcb154" + }, + { + "bytes": 31035526, + "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/test_batch", + "sha256": "f53d8d457504f7cff4ea9e021afcf0e0ad8e24a91f3fc42091b8adef61157831" + } + ], + "dataset": "cifar10", + "evaluation_split": "validation", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "random_crop_32_padding_4_zero_then_horizontal_flip_p0.5", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + }, + "timing": { + "apical_warmup_wall_s": 5.97963809967041, + "evaluation_wall_s": 0.3711128234863281, + "predictor_warmup_wall_s": 0.0, + "timing_excludes_data_loading_hashing_and_model_construction": true, + "total_timed_wall_s": 6.350949764251709, + "train_wall_s": 0.0 + }, + "work": { + "apical_macs_per_example": 1884160, + "causal_scalar_observations": 200, + "components": { + "apical_projection_macs": 23906222080, + "apical_regression_macs": 23906222080, + "bp_reverse_macs_estimate": 0, + "local_weight_correlation_macs": 0, + "ordinary_forward_macs": 0, + "perturbation_forward_macs": 1029023191040, + "warmup_clean_forward_macs": 514511595520 + }, + "definition": "multiply-accumulates in conv/linear maps; one local weight correlation equals one forward-weight MAC count; BP reverse is estimated as one weight-gradient plus one activation-gradient convolution per forward convolution; elementwise nonlinearities and optimizer arithmetic excluded", + "forward_macs_per_example": 40551040, + "logical_batch_loss_queries": 200, + "per_example_cross_entropy_terms": 25376, + "total_clean_forward_examples": 12688, + "total_forward_equivalent_examples": 38064, + "total_macs_estimate": 1591347230720 + } +} diff --git a/results/oral_a_apical_selection.json b/results/oral_a_apical_selection.json new file mode 100644 index 0000000..fd5a90e --- /dev/null +++ b/results/oral_a_apical_selection.json @@ -0,0 +1,219 @@ +{ + "confirmation_test_seeds_touched": false, + "protocol": "oral_a_A2a_v1", + "rows": [ + { + "a_scale": 0.0, + "eligible": false, + "eta_A": 0.001, + "metrics": { + "all_layer_alignment": 0.0022820909017402875, + "calibration_mse": 518.2959037449049, + "early_third_alignment": -0.0007199425793563327, + "prediction_target_cosine": -2.7486039723463496e-05 + }, + "path": "results/oral_a_apical_screen/channel_gated_a0_etaA0.001.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "channel_gated", + "vectorizer_parameters": 13760 + }, + { + "a_scale": 0.0, + "eligible": false, + "eta_A": 0.01, + "metrics": { + "all_layer_alignment": 0.002390206068460094, + "calibration_mse": 518.2959435504416, + "early_third_alignment": -0.0006589657083774606, + "prediction_target_cosine": -3.4022074583332044e-05 + }, + "path": "results/oral_a_apical_screen/channel_gated_a0_etaA0.01.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "channel_gated", + "vectorizer_parameters": 13760 + }, + { + "a_scale": 0.0, + "eligible": false, + "eta_A": 0.1, + "metrics": { + "all_layer_alignment": 0.0029464198496976964, + "calibration_mse": 518.2965393066406, + "early_third_alignment": -9.475171100348234e-05, + "prediction_target_cosine": -7.547625140869704e-05 + }, + "path": "results/oral_a_apical_screen/channel_gated_a0_etaA0.1.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "channel_gated", + "vectorizer_parameters": 13760 + }, + { + "a_scale": 1.0, + "eligible": true, + "eta_A": 0.001, + "metrics": { + "all_layer_alignment": -0.0031654090822772368, + "calibration_mse": 518.2959050717561, + "early_third_alignment": 0.0014168331302547206, + "prediction_target_cosine": 0.00016553175127441687 + }, + "path": "results/oral_a_apical_screen/channel_gated_a1_etaA0.001.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "channel_gated", + "vectorizer_parameters": 13760 + }, + { + "a_scale": 1.0, + "eligible": false, + "eta_A": 0.01, + "metrics": { + "all_layer_alignment": 0.000495402203676732, + "calibration_mse": 518.2959276282269, + "early_third_alignment": -0.00027415525012960035, + "prediction_target_cosine": 3.61075533430761e-05 + }, + "path": "results/oral_a_apical_screen/channel_gated_a1_etaA0.01.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "channel_gated", + "vectorizer_parameters": 13760 + }, + { + "a_scale": 1.0, + "eligible": false, + "eta_A": 0.1, + "metrics": { + "all_layer_alignment": 0.0028739075396994225, + "calibration_mse": 518.296547267748, + "early_third_alignment": -7.813629539062579e-05, + "prediction_target_cosine": -7.364532238643221e-05 + }, + "path": "results/oral_a_apical_screen/channel_gated_a1_etaA0.1.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "channel_gated", + "vectorizer_parameters": 13760 + }, + { + "a_scale": 0.0, + "eligible": true, + "eta_A": 0.001, + "metrics": { + "all_layer_alignment": 0.0017333623683570248, + "calibration_mse": 518.2958984375, + "early_third_alignment": 8.52438776443402e-05, + "prediction_target_cosine": 0.0001586056141805217 + }, + "path": "results/oral_a_apical_screen/spatial_template_a0_etaA0.001.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "spatial_template", + "vectorizer_parameters": 1884160 + }, + { + "a_scale": 0.0, + "eligible": true, + "eta_A": 0.01, + "metrics": { + "all_layer_alignment": 0.0017124206091179268, + "calibration_mse": 518.2982058317765, + "early_third_alignment": 5.210153176449239e-05, + "prediction_target_cosine": 0.00015677468540085558 + }, + "path": "results/oral_a_apical_screen/spatial_template_a0_etaA0.01.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "spatial_template", + "vectorizer_parameters": 1884160 + }, + { + "a_scale": 0.0, + "eligible": false, + "eta_A": 0.1, + "metrics": { + "all_layer_alignment": 0.0014395472229041747, + "calibration_mse": 518.4102902619735, + "early_third_alignment": -0.00022858582572856298, + "prediction_target_cosine": 0.00016023495648166922 + }, + "path": "results/oral_a_apical_screen/spatial_template_a0_etaA0.1.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "spatial_template", + "vectorizer_parameters": 1884160 + }, + { + "a_scale": 1.0, + "eligible": true, + "eta_A": 0.001, + "metrics": { + "all_layer_alignment": 0.0016821367041158833, + "calibration_mse": 518.2959063986074, + "early_third_alignment": 2.9081663039202493e-05, + "prediction_target_cosine": 0.00011094105930582137 + }, + "path": "results/oral_a_apical_screen/spatial_template_a1_etaA0.001.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "spatial_template", + "vectorizer_parameters": 1884160 + }, + { + "a_scale": 1.0, + "eligible": true, + "eta_A": 0.01, + "metrics": { + "all_layer_alignment": 0.0017294948277259735, + "calibration_mse": 518.2982217539911, + "early_third_alignment": 4.663398916212221e-05, + "prediction_target_cosine": 0.0001529906778155134 + }, + "path": "results/oral_a_apical_screen/spatial_template_a1_etaA0.01.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "spatial_template", + "vectorizer_parameters": 1884160 + }, + { + "a_scale": 1.0, + "eligible": false, + "eta_A": 0.1, + "metrics": { + "all_layer_alignment": 0.0014408246727390705, + "calibration_mse": 518.4102889351223, + "early_third_alignment": -0.00022893324785400182, + "prediction_target_cosine": 0.00016010901837600736 + }, + "path": "results/oral_a_apical_screen/spatial_template_a1_etaA0.1.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "spatial_template", + "vectorizer_parameters": 1884160 + } + ], + "selected": { + "channel_gated": { + "a_scale": 1.0, + "eligible": true, + "eta_A": 0.001, + "metrics": { + "all_layer_alignment": -0.0031654090822772368, + "calibration_mse": 518.2959050717561, + "early_third_alignment": 0.0014168331302547206, + "prediction_target_cosine": 0.00016553175127441687 + }, + "path": "results/oral_a_apical_screen/channel_gated_a1_etaA0.001.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "channel_gated", + "vectorizer_parameters": 13760 + }, + "spatial_template": { + "a_scale": 0.0, + "eligible": true, + "eta_A": 0.001, + "metrics": { + "all_layer_alignment": 0.0017333623683570248, + "calibration_mse": 518.2958984375, + "early_third_alignment": 8.52438776443402e-05, + "prediction_target_cosine": 0.0001586056141805217 + }, + "path": "results/oral_a_apical_screen/spatial_template_a0_etaA0.001.json", + "source_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", + "vectorizer_mode": "spatial_template", + "vectorizer_parameters": 1884160 + } + }, + "status": "selected" +} |
