{ "apical_warmup": { "last": { "calibration_mse": 290.18821185568106, "prediction_target_cosine": 0.00037572678598497784, "target_power": 290.188232421875 }, "steps": 100 }, "architecture": { "adaptive_apical_parameters": 390592, "base_width": 16, "blocks_per_stage": 3, "bn_eps": 1e-05, "bn_momentum": 0.1, "depth": 20, "family": "CIFAR 6n+2 ResNet, option-A shortcuts", "fixed_traffic_coefficients": 188416, "forward_parameters": 269722, "hidden_shapes": [ [ 16, 32, 32 ], [ 16, 32, 32 ], [ 16, 32, 32 ], [ 16, 32, 32 ], [ 16, 32, 32 ], [ 16, 32, 32 ], [ 16, 32, 32 ], [ 32, 16, 16 ], [ 32, 16, 16 ], [ 32, 16, 16 ], [ 32, 16, 16 ], [ 32, 16, 16 ], [ 32, 16, 16 ], [ 64, 8, 8 ], [ 64, 8, 8 ], [ 64, 8, 8 ], [ 64, 8, 8 ], [ 64, 8, 8 ], [ 64, 8, 8 ] ], "normalization": "batchnorm", "predictor_parameters": 376832, "residual_scale": 1.0, "vectorizer_mode": "channel_gated", "vectorizer_parameters": 13760 }, "args": { "a_scale": 1.0, "a_warmup_steps": 100, "alignment_probe": 32, "apical_seed": null, "augment_train": 1, "batch_size": 128, "bn_eps": 1e-05, "bn_momentum": 0.1, "data_dir": "/home/yurenh2/sdrn/data", "depth": 20, "device": "cuda:0", "epochs": 200, "eta_A": 0.001, "eta_P": 0.01, "eval_every": 20, "eval_split": "validation", "learn_P": 0, "loader_seed": 0, "lr": 0.03, "lr_gamma": 0.1, "lr_milestones": "100,150", "lr_schedule": "step", "max_steps": 0, "mode": "sdil", "momentum": 0.9, "normalization": "batchnorm", "nuisance_scale": 0.0, "out": "results/oral_a_dev/sdil_full_r20_s0.json", "output_lr": 0.1, "pert_directions": 1, "pert_every": 4, "pert_sigma": 0.01, "perturb_seed": 1000, "predictor_warmup_steps": 0, "residual_scale": null, "seed": 0, "split_seed": 2027, "train_limit": 0, "use_residual": 1, "val_examples": 5000, "vectorizer_mode": "channel_gated", "warmup_epochs": 0, "weight_decay": 0.0001, "weight_scale": 1.0, "width": 16 }, "counters": { "apical_warmup_examples": 12800, "calibration_event_examples": 2265600, "causal_scalar_observations": 35400, "logical_batch_loss_queries": 35400, "ordinary_examples": 9000000, "per_example_loss_terms": 4531200, "perturbation_events": 17700, "perturbation_forward_examples": 4531200, "predictor_warmup_examples": 0 }, "diagnostics": { "early_third_mean": NaN, "innovation_negative_gradient_cosine": [ NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN ], "normalization_state": "training_batch_stats_without_running_update", "raw_negative_gradient_cosine": [ NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN ], "teaching_negative_gradient_cosine": [ NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN, NaN ], "wall_s": 0.12295675277709961 }, "epochs": [ { "calibration": { "calibration_mse": 142.26092846940978, "prediction_target_cosine": -5.120151685872971e-06, "target_power": 142.2609250246797 }, "epoch": 1, "lr": 0.03, "output_lr": 0.1, "step": 352, "train_examples": 45000, "train_loss": 2.4248756050109863 }, { "calibration": { "calibration_mse": 170.77780406192872, "prediction_target_cosine": 2.130938988042182e-05, "target_power": 170.77779697238105 }, "epoch": 2, "lr": 0.03, "output_lr": 0.1, "step": 704, "train_examples": 45000, "train_loss": 2.3075676691691083 }, { "calibration": { "calibration_mse": 189.64478362977627, "prediction_target_cosine": 1.9699343209067078e-05, "target_power": 189.64478348164448 }, "epoch": 3, "lr": 0.03, "output_lr": 0.1, "step": 1056, "train_examples": 45000, "train_loss": 2.2274199758105806 }, { "calibration": { "calibration_mse": 195.31212744229245, "prediction_target_cosine": 1.3287171579791927e-05, "target_power": 195.3121247766815 }, "epoch": 4, "lr": 0.03, "output_lr": 0.1, "step": 1408, "train_examples": 45000, "train_loss": 2.236113543404473 }, { "calibration": { "calibration_mse": 131.69457619605973, "prediction_target_cosine": -1.9181219483840043e-05, "target_power": 131.6945734072933 }, "epoch": 5, "lr": 0.03, "output_lr": 0.1, "step": 1760, "train_examples": 45000, "train_loss": 2.1845484929402668 }, { "calibration": { "calibration_mse": 214.62042681660435, "prediction_target_cosine": 1.5224048789643175e-05, "target_power": 214.62042582989946 }, "epoch": 6, "lr": 0.03, "output_lr": 0.1, "step": 2112, "train_examples": 45000, "train_loss": 2.2274387500339086 }, { "calibration": { "calibration_mse": 249.44172076644045, "prediction_target_cosine": 4.735744266223287e-06, "target_power": 249.4417245881502 }, "epoch": 7, "lr": 0.03, "output_lr": 0.1, "step": 2464, "train_examples": 45000, "train_loss": 2.1511679939058093 }, { "calibration": { "calibration_mse": 241.9874178501962, "prediction_target_cosine": 2.871340625625777e-05, "target_power": 241.98741854901007 }, "epoch": 8, "lr": 0.03, "output_lr": 0.1, "step": 2816, "train_examples": 45000, "train_loss": 2.3858975699106852 }, { "calibration": { "calibration_mse": 230.36731525985041, "prediction_target_cosine": 2.7813335307394514e-05, "target_power": 230.36731070333818 }, "epoch": 9, "lr": 0.03, "output_lr": 0.1, "step": 3168, "train_examples": 45000, "train_loss": 2.324350842836168 }, { "calibration": { "calibration_mse": 261.16900031712544, "prediction_target_cosine": 4.56942525344397e-05, "target_power": 261.1689959096285 }, "epoch": 10, "lr": 0.03, "output_lr": 0.1, "step": 3520, "train_examples": 45000, "train_loss": 2.4430940684424507 }, { "calibration": { "calibration_mse": 362.4051081771026, "prediction_target_cosine": 1.7848043967156622e-05, "target_power": 362.4051154071001 }, "epoch": 11, "lr": 0.03, "output_lr": 0.1, "step": 3872, "train_examples": 45000, "train_loss": 3.116123947270711 }, { "calibration": { "calibration_mse": 315.3471581766424, "prediction_target_cosine": 7.536274067643723e-05, "target_power": 315.34715614577004 }, "epoch": 12, "lr": 0.03, "output_lr": 0.1, "step": 4224, "train_examples": 45000, "train_loss": 3.264699432796902 }, { "calibration": { "calibration_mse": 356.58399579827153, "prediction_target_cosine": 7.467543063446071e-05, "target_power": 356.5839798142434 }, "epoch": 13, "lr": 0.03, "output_lr": 0.1, "step": 4576, "train_examples": 45000, "train_loss": 3.3854243429395887 }, { "calibration": { "calibration_mse": 404.3220514495382, "prediction_target_cosine": 1.593798731590023e-05, "target_power": 404.3220492818434 }, "epoch": 14, "lr": 0.03, "output_lr": 0.1, "step": 4928, "train_examples": 45000, "train_loss": 3.636718994225396 }, { "calibration": { "calibration_mse": 352.03641794218316, "prediction_target_cosine": 6.031522331721257e-05, "target_power": 352.0364126960632 }, "epoch": 15, "lr": 0.03, "output_lr": 0.1, "step": 5280, "train_examples": 45000, "train_loss": 3.9225536643981935 }, { "calibration": { "calibration_mse": 383.77995509406725, "prediction_target_cosine": -2.535504656905698e-05, "target_power": 383.77994124611956 }, "epoch": 16, "lr": 0.03, "output_lr": 0.1, "step": 5632, "train_examples": 45000, "train_loss": 3.6204674806806776 }, { "calibration": { "calibration_mse": 453.1726072217989, "prediction_target_cosine": 3.8347924248536074e-05, "target_power": 453.17260394038516 }, "epoch": 17, "lr": 0.03, "output_lr": 0.1, "step": 5984, "train_examples": 45000, "train_loss": 3.735507978439331 }, { "calibration": { "calibration_mse": 494.781739041651, "prediction_target_cosine": 1.6093348730532647e-05, "target_power": 494.7817316142686 }, "epoch": 18, "lr": 0.03, "output_lr": 0.1, "step": 6336, "train_examples": 45000, "train_loss": 4.134002408599853 }, { "calibration": { "calibration_mse": 529.3938995664971, "prediction_target_cosine": 4.966474897062738e-05, "target_power": 529.3938993744242 }, "epoch": 19, "lr": 0.03, "output_lr": 0.1, "step": 6688, "train_examples": 45000, "train_loss": 4.730613648139106 }, { "calibration": { "calibration_mse": 492.22622147554415, "prediction_target_cosine": -9.188050003749003e-06, "target_power": 492.22621874637883 }, "epoch": 20, "eval_accuracy": 0.3526, "eval_loss": 5.06778037109375, "eval_split": "validation", "lr": 0.03, "output_lr": 0.1, "step": 7040, "train_examples": 45000, "train_loss": 4.28790976164076 }, { "calibration": { "calibration_mse": 631.5754133367934, "prediction_target_cosine": 4.641877463523035e-05, "target_power": 631.5754141588454 }, "epoch": 21, "lr": 0.03, "output_lr": 0.1, "step": 7392, "train_examples": 45000, "train_loss": 4.931362113444011 }, { "calibration": { "calibration_mse": 705.0312621907254, "prediction_target_cosine": 1.5874308489431493e-06, "target_power": 705.0312680084065 }, "epoch": 22, "lr": 0.03, "output_lr": 0.1, "step": 7744, "train_examples": 45000, "train_loss": 6.228103899637858 }, { "calibration": { "calibration_mse": 869.3405450967323, "prediction_target_cosine": 5.1928815228082084e-05, "target_power": 869.3405314628545 }, "epoch": 23, "lr": 0.03, "output_lr": 0.1, "step": 8096, "train_examples": 45000, "train_loss": 6.058901812065972 }, { "calibration": { "calibration_mse": 1030.0436134480512, "prediction_target_cosine": -1.108869497719368e-05, "target_power": 1030.0436205758303 }, "epoch": 24, "lr": 0.03, "output_lr": 0.1, "step": 8448, "train_examples": 45000, "train_loss": 6.3764598000420465 }, { "calibration": { "calibration_mse": 875.4920511692325, "prediction_target_cosine": 5.024974477863364e-05, "target_power": 875.4920396014695 }, "epoch": 25, "lr": 0.03, "output_lr": 0.1, "step": 8800, "train_examples": 45000, "train_loss": 6.072056756337484 }, { "calibration": { "calibration_mse": 1216.320880951809, "prediction_target_cosine": 3.563140890986608e-05, "target_power": 1216.320865344337 }, "epoch": 26, "lr": 0.03, "output_lr": 0.1, "step": 9152, "train_examples": 45000, "train_loss": 6.714967920769586 }, { "calibration": { "calibration_mse": 1006.585065626843, "prediction_target_cosine": 1.5808554377597195e-06, "target_power": 1006.5850547987287 }, "epoch": 27, "lr": 0.03, "output_lr": 0.1, "step": 9504, "train_examples": 45000, "train_loss": 6.634224433220758 }, { "calibration": { "calibration_mse": 1293.9108654121094, "prediction_target_cosine": 3.434976504469591e-05, "target_power": 1293.910856750407 }, "epoch": 28, "lr": 0.03, "output_lr": 0.1, "step": 9856, "train_examples": 45000, "train_loss": 7.361039043850369 }, { "calibration": { "calibration_mse": 1168.6127594745396, "prediction_target_cosine": 2.05039191501232e-05, "target_power": 1168.6127624916053 }, "epoch": 29, "lr": 0.03, "output_lr": 0.1, "step": 10208, "train_examples": 45000, "train_loss": 6.89228710776435 }, { "calibration": { "calibration_mse": 1217.4179927003038, "prediction_target_cosine": 4.899155528236468e-05, "target_power": 1217.4179925077221 }, "epoch": 30, "lr": 0.03, "output_lr": 0.1, "step": 10560, "train_examples": 45000, "train_loss": 7.254968173641629 }, { "calibration": { "calibration_mse": 1323.2104840733284, "prediction_target_cosine": 1.2728994554210635e-05, "target_power": 1323.2104647397446 }, "epoch": 31, "lr": 0.03, "output_lr": 0.1, "step": 10912, "train_examples": 45000, "train_loss": 7.389020330980089 }, { "calibration": { "calibration_mse": 1377.2780309310072, "prediction_target_cosine": 4.572166737244789e-06, "target_power": 1377.2780412487361 }, "epoch": 32, "lr": 0.03, "output_lr": 0.1, "step": 11264, "train_examples": 45000, "train_loss": 8.22750666724311 }, { "calibration": { "calibration_mse": 1521.2079438092298, "prediction_target_cosine": 3.9589531839474834e-05, "target_power": 1521.2079332335866 }, "epoch": 33, "lr": 0.03, "output_lr": 0.1, "step": 11616, "train_examples": 45000, "train_loss": 8.618234800889757 }, { "calibration": { "calibration_mse": 1714.481807423439, "prediction_target_cosine": 6.564467478941335e-05, "target_power": 1714.481808286175 }, "epoch": 34, "lr": 0.03, "output_lr": 0.1, "step": 11968, "train_examples": 45000, "train_loss": 9.987927119106716 }, { "calibration": { "calibration_mse": 1286.147588690169, "prediction_target_cosine": 5.3948176124196105e-05, "target_power": 1286.1475643246129 }, "epoch": 35, "lr": 0.03, "output_lr": 0.1, "step": 12320, "train_examples": 45000, "train_loss": 10.262047458733452 }, { "calibration": { "calibration_mse": 1889.400973652684, "prediction_target_cosine": 3.526118696767705e-05, "target_power": 1889.4009319308236 }, "epoch": 36, "lr": 0.03, "output_lr": 0.1, "step": 12672, "train_examples": 45000, "train_loss": 12.48206615854899 }, { "calibration": { "calibration_mse": 2847.7376760077254, "prediction_target_cosine": 5.890126577784353e-05, "target_power": 2847.737692615025 }, "epoch": 37, "lr": 0.03, "output_lr": 0.1, "step": 13024, "train_examples": 45000, "train_loss": 11.426354922146267 }, { "calibration": { "calibration_mse": 2254.2651952857327, "prediction_target_cosine": 3.57443530697733e-05, "target_power": 2254.265174126908 }, "epoch": 38, "lr": 0.03, "output_lr": 0.1, "step": 13376, "train_examples": 45000, "train_loss": 14.126089464314779 }, { "calibration": { "calibration_mse": 3570.404625115017, "prediction_target_cosine": 1.6840951784089786e-05, "target_power": 3570.404574593537 }, "epoch": 39, "lr": 0.03, "output_lr": 0.1, "step": 13728, "train_examples": 45000, "train_loss": 15.35085629848904 }, { "calibration": { "calibration_mse": 3347.859104873937, "prediction_target_cosine": -1.6028913891638828e-05, "target_power": 3347.859035403952 }, "epoch": 40, "eval_accuracy": 0.2844, "eval_loss": 17.8836359375, "eval_split": "validation", "lr": 0.03, "output_lr": 0.1, "step": 14080, "train_examples": 45000, "train_loss": 16.475267074924044 }, { "calibration": { "calibration_mse": 3620.2920621072526, "prediction_target_cosine": 1.7698698598729218e-05, "target_power": 3620.2919751598897 }, "epoch": 41, "lr": 0.03, "output_lr": 0.1, "step": 14432, "train_examples": 45000, "train_loss": 18.26095764126248 }, { "calibration": { "calibration_mse": 3372.267064184442, "prediction_target_cosine": 4.054439075274129e-05, "target_power": 3372.2671280799823 }, "epoch": 42, "lr": 0.03, "output_lr": 0.1, "step": 14784, "train_examples": 45000, "train_loss": 18.189921793789335 }, { "calibration": { "calibration_mse": 5057.76865694964, "prediction_target_cosine": 3.154939702252931e-05, "target_power": 5057.768832532961 }, "epoch": 43, "lr": 0.03, "output_lr": 0.1, "step": 15136, "train_examples": 45000, "train_loss": 20.16826618448893 }, { "calibration": { "calibration_mse": 6378.545264090003, "prediction_target_cosine": 1.666679367705587e-05, "target_power": 6378.545273705433 }, "epoch": 44, "lr": 0.03, "output_lr": 0.1, "step": 15488, "train_examples": 45000, "train_loss": 21.47305562693278 }, { "calibration": { "calibration_mse": 3860.1069145137853, "prediction_target_cosine": 4.1874467305640655e-05, "target_power": 3860.1068515447773 }, "epoch": 45, "lr": 0.03, "output_lr": 0.1, "step": 15840, "train_examples": 45000, "train_loss": 23.228443602159288 }, { "calibration": { "calibration_mse": 4111.717137232277, "prediction_target_cosine": 2.0983340366496623e-05, "target_power": 4111.717064274953 }, "epoch": 46, "lr": 0.03, "output_lr": 0.1, "step": 16192, "train_examples": 45000, "train_loss": 26.718320481363932 }, { "calibration": { "calibration_mse": 6623.266515922787, "prediction_target_cosine": 3.414601460981923e-05, "target_power": 6623.266473941065 }, "epoch": 47, "lr": 0.03, "output_lr": 0.1, "step": 16544, "train_examples": 45000, "train_loss": 26.138589851548936 }, { "calibration": { "calibration_mse": 6087.71189168614, "prediction_target_cosine": 3.873974129150477e-05, "target_power": 6087.711752356744 }, "epoch": 48, "lr": 0.03, "output_lr": 0.1, "step": 16896, "train_examples": 45000, "train_loss": 30.99675353325738 }, { "calibration": { "calibration_mse": 7202.803725531746, "prediction_target_cosine": 7.462676452085615e-05, "target_power": 7202.803723873826 }, "epoch": 49, "lr": 0.03, "output_lr": 0.1, "step": 17248, "train_examples": 45000, "train_loss": 28.21036580980089 }, { "calibration": { "calibration_mse": 6134.1136319374145, "prediction_target_cosine": 4.322624471257815e-05, "target_power": 6134.113787830583 }, "epoch": 50, "lr": 0.03, "output_lr": 0.1, "step": 17600, "train_examples": 45000, "train_loss": 29.816331009928387 }, { "calibration": { "calibration_mse": 6723.329940029519, "prediction_target_cosine": 3.244871155240453e-05, "target_power": 6723.330072100454 }, "epoch": 51, "lr": 0.03, "output_lr": 0.1, "step": 17952, "train_examples": 45000, "train_loss": 29.147723817952475 }, { "calibration": { "calibration_mse": 7977.255477374929, "prediction_target_cosine": 3.6955415995607416e-06, "target_power": 7977.255535312789 }, "epoch": 52, "lr": 0.03, "output_lr": 0.1, "step": 18304, "train_examples": 45000, "train_loss": 34.91372113986545 }, { "calibration": { "calibration_mse": 10323.407597489067, "prediction_target_cosine": -9.126162131415332e-06, "target_power": 10323.407674211876 }, "epoch": 53, "lr": 0.03, "output_lr": 0.1, "step": 18656, "train_examples": 45000, "train_loss": 37.32762109103733 }, { "calibration": { "calibration_mse": 9520.888989746127, "prediction_target_cosine": 3.946130177208881e-05, "target_power": 9520.888864434637 }, "epoch": 54, "lr": 0.03, "output_lr": 0.1, "step": 19008, "train_examples": 45000, "train_loss": 36.95214984266493 }, { "calibration": { "calibration_mse": 10012.82621979749, "prediction_target_cosine": 4.050476833716903e-05, "target_power": 10012.826482390583 }, "epoch": 55, "lr": 0.03, "output_lr": 0.1, "step": 19360, "train_examples": 45000, "train_loss": 39.117197730848524 }, { "calibration": { "calibration_mse": 12023.639591636726, "prediction_target_cosine": 2.3378804475242e-06, "target_power": 12023.639589814331 }, "epoch": 56, "lr": 0.03, "output_lr": 0.1, "step": 19712, "train_examples": 45000, "train_loss": 42.929720042588976 }, { "calibration": { "calibration_mse": 13326.782062088076, "prediction_target_cosine": 5.352798341532508e-05, "target_power": 13326.782067844993 }, "epoch": 57, "lr": 0.03, "output_lr": 0.1, "step": 20064, "train_examples": 45000, "train_loss": 44.34269226142035 }, { "calibration": { "calibration_mse": 12457.90936806135, "prediction_target_cosine": 3.586177762983611e-05, "target_power": 12457.909191807737 }, "epoch": 58, "lr": 0.03, "output_lr": 0.1, "step": 20416, "train_examples": 45000, "train_loss": 53.25831157972548 }, { "calibration": { "calibration_mse": 14505.218685153437, "prediction_target_cosine": -9.702282143099002e-06, "target_power": 14505.218814834778 }, "epoch": 59, "lr": 0.03, "output_lr": 0.1, "step": 20768, "train_examples": 45000, "train_loss": 56.738197840711805 }, { "calibration": { "calibration_mse": 12104.714603378052, "prediction_target_cosine": 4.038185739212611e-05, "target_power": 12104.714606234214 }, "epoch": 60, "eval_accuracy": 0.3084, "eval_loss": 48.81988203125, "eval_split": "validation", "lr": 0.03, "output_lr": 0.1, "step": 21120, "train_examples": 45000, "train_loss": 61.89007039523654 }, { "calibration": { "calibration_mse": 16957.135531766136, "prediction_target_cosine": 6.391218994340552e-05, "target_power": 16957.13540488458 }, "epoch": 61, "lr": 0.03, "output_lr": 0.1, "step": 21472, "train_examples": 45000, "train_loss": 66.32816502956814 }, { "calibration": { "calibration_mse": 23589.800162002968, "prediction_target_cosine": 3.869866879570572e-05, "target_power": 23589.800378499753 }, "epoch": 62, "lr": 0.03, "output_lr": 0.1, "step": 21824, "train_examples": 45000, "train_loss": 66.4304635281033 }, { "calibration": { "calibration_mse": 17063.567628094486, "prediction_target_cosine": 1.2365923320404936e-05, "target_power": 17063.567225467075 }, "epoch": 63, "lr": 0.03, "output_lr": 0.1, "step": 22176, "train_examples": 45000, "train_loss": 61.848114156765405 }, { "calibration": { "calibration_mse": 20145.622601572708, "prediction_target_cosine": 4.975189064002332e-05, "target_power": 20145.62256209558 }, "epoch": 64, "lr": 0.03, "output_lr": 0.1, "step": 22528, "train_examples": 45000, "train_loss": 63.27609741210937 }, { "calibration": { "calibration_mse": 23949.849978939816, "prediction_target_cosine": 5.491976438560331e-05, "target_power": 23949.849776010742 }, "epoch": 65, "lr": 0.03, "output_lr": 0.1, "step": 22880, "train_examples": 45000, "train_loss": 70.30348611246745 }, { "calibration": { "calibration_mse": 23350.559424263887, "prediction_target_cosine": 3.549862288294985e-05, "target_power": 23350.55949317043 }, "epoch": 66, "lr": 0.03, "output_lr": 0.1, "step": 23232, "train_examples": 45000, "train_loss": 76.89770236002605 }, { "calibration": { "calibration_mse": 24210.404024987594, "prediction_target_cosine": 3.108741730684027e-05, "target_power": 24210.403118962826 }, "epoch": 67, "lr": 0.03, "output_lr": 0.1, "step": 23584, "train_examples": 45000, "train_loss": 108.41309141710069 }, { "calibration": { "calibration_mse": 23079.73656247355, "prediction_target_cosine": -3.70013756328733e-05, "target_power": 23079.735782427273 }, "epoch": 68, "lr": 0.03, "output_lr": 0.1, "step": 23936, "train_examples": 45000, "train_loss": 116.28849081217447 }, { "calibration": { "calibration_mse": 51396.446248936576, "prediction_target_cosine": 8.923274095527056e-06, "target_power": 51396.44616932927 }, "epoch": 69, "lr": 0.03, "output_lr": 0.1, "step": 24288, "train_examples": 45000, "train_loss": 143.1851363172743 }, { "calibration": { "calibration_mse": 45116.00593224062, "prediction_target_cosine": 1.567150501795464e-05, "target_power": 45116.00633214868 }, "epoch": 70, "lr": 0.03, "output_lr": 0.1, "step": 24640, "train_examples": 45000, "train_loss": 146.16235985243057 }, { "calibration": { "calibration_mse": 63617.30153912576, "prediction_target_cosine": 9.694315162471824e-07, "target_power": 63617.30126043931 }, "epoch": 71, "lr": 0.03, "output_lr": 0.1, "step": 24992, "train_examples": 45000, "train_loss": 167.29566240505642 }, { "calibration": { "calibration_mse": 59029.095404237036, "prediction_target_cosine": -3.8879670438084225e-06, "target_power": 59029.095730216126 }, "epoch": 72, "lr": 0.03, "output_lr": 0.1, "step": 25344, "train_examples": 45000, "train_loss": 201.25478683946397 }, { "calibration": { "calibration_mse": 93548.83073081424, "prediction_target_cosine": 4.02229294978188e-05, "target_power": 93548.83068157091 }, "epoch": 73, "lr": 0.03, "output_lr": 0.1, "step": 25696, "train_examples": 45000, "train_loss": 255.2457300889757 }, { "calibration": { "calibration_mse": 125623.44832417759, "prediction_target_cosine": 4.708052744506203e-05, "target_power": 125623.44980211217 }, "epoch": 74, "lr": 0.03, "output_lr": 0.1, "step": 26048, "train_examples": 45000, "train_loss": 343.87444676649307 }, { "calibration": { "calibration_mse": 124640.97399765471, "prediction_target_cosine": -5.064306365835342e-06, "target_power": 124640.97530615279 }, "epoch": 75, "lr": 0.03, "output_lr": 0.1, "step": 26400, "train_examples": 45000, "train_loss": 360.54264365234377 }, { "calibration": { "calibration_mse": 241851.34037842014, "prediction_target_cosine": 7.968131389886469e-05, "target_power": 241851.3417864261 }, "epoch": 76, "lr": 0.03, "output_lr": 0.1, "step": 26752, "train_examples": 45000, "train_loss": 398.3567958170573 }, { "calibration": { "calibration_mse": 199514.99018743765, "prediction_target_cosine": 1.9350905005203684e-05, "target_power": 199514.99037634814 }, "epoch": 77, "lr": 0.03, "output_lr": 0.1, "step": 27104, "train_examples": 45000, "train_loss": 469.2747630533854 }, { "calibration": { "calibration_mse": 162271.13078702174, "prediction_target_cosine": 2.5799827295987502e-05, "target_power": 162271.1310087966 }, "epoch": 78, "lr": 0.03, "output_lr": 0.1, "step": 27456, "train_examples": 45000, "train_loss": 518.902432779948 }, { "calibration": { "calibration_mse": 329178.31447980023, "prediction_target_cosine": 3.2362186994689176e-05, "target_power": 329178.313448186 }, "epoch": 79, "lr": 0.03, "output_lr": 0.1, "step": 27808, "train_examples": 45000, "train_loss": 620.2860517686632 }, { "calibration": { "calibration_mse": 364280.45720910426, "prediction_target_cosine": 6.468792391264821e-05, "target_power": 364280.45838307025 }, "epoch": 80, "eval_accuracy": 0.2282, "eval_loss": 1080.0792, "eval_split": "validation", "lr": 0.03, "output_lr": 0.1, "step": 28160, "train_examples": 45000, "train_loss": 666.1459060112848 }, { "calibration": { "calibration_mse": 560786.7693639989, "prediction_target_cosine": 5.0133754272917375e-05, "target_power": 560786.7724463422 }, "epoch": 81, "lr": 0.03, "output_lr": 0.1, "step": 28512, "train_examples": 45000, "train_loss": 852.9150838324653 }, { "calibration": { "calibration_mse": 448151.9335011955, "prediction_target_cosine": 2.266160576744399e-05, "target_power": 448151.9326951098 }, "epoch": 82, "lr": 0.03, "output_lr": 0.1, "step": 28864, "train_examples": 45000, "train_loss": 789.5596756944444 }, { "calibration": { "calibration_mse": 570399.1552796762, "prediction_target_cosine": 2.8400752579445593e-05, "target_power": 570399.1497913705 }, "epoch": 83, "lr": 0.03, "output_lr": 0.1, "step": 29216, "train_examples": 45000, "train_loss": 839.9816970811632 }, { "calibration": { "calibration_mse": 1514968.9443011982, "prediction_target_cosine": 3.440886662575942e-05, "target_power": 1514968.9244389588 }, "epoch": 84, "lr": 0.03, "output_lr": 0.1, "step": 29568, "train_examples": 45000, "train_loss": 963.7592224826388 }, { "calibration": { "calibration_mse": 5975441.481893999, "prediction_target_cosine": 8.508675264290486e-05, "target_power": 5975441.54251376 }, "epoch": 85, "lr": 0.03, "output_lr": 0.1, "step": 29920, "train_examples": 45000, "train_loss": 1105.7151870008681 }, { "calibration": { "calibration_mse": 54027101.58410685, "prediction_target_cosine": -7.730766330982968e-06, "target_power": 54027105.346407585 }, "epoch": 86, "lr": 0.03, "output_lr": 0.1, "step": 30272, "train_examples": 45000, "train_loss": 2679.8837784722223 }, { "calibration": { "calibration_mse": 302564393899.4321, "prediction_target_cosine": 1.4983485078330897e-05, "target_power": 302564383414.18854 }, "epoch": 87, "lr": 0.03, "output_lr": 0.1, "step": 30624, "train_examples": 45000, "train_loss": 35881231.129456945 }, { "calibration": { "calibration_mse": 324392029489.5691, "prediction_target_cosine": -3.0267985399461017e-05, "target_power": 324392029090.9091 }, "epoch": 88, "lr": 0.03, "output_lr": 0.1, "step": 30976, "train_examples": 45000, "train_loss": 62859708.14577778 }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 89, "lr": 0.03, "output_lr": 0.1, "step": 31328, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 90, "lr": 0.03, "output_lr": 0.1, "step": 31680, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 91, "lr": 0.03, "output_lr": 0.1, "step": 32032, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 92, "lr": 0.03, "output_lr": 0.1, "step": 32384, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 93, "lr": 0.03, "output_lr": 0.1, "step": 32736, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 94, "lr": 0.03, "output_lr": 0.1, "step": 33088, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 95, "lr": 0.03, "output_lr": 0.1, "step": 33440, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 96, "lr": 0.03, "output_lr": 0.1, "step": 33792, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 97, "lr": 0.03, "output_lr": 0.1, "step": 34144, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 98, "lr": 0.03, "output_lr": 0.1, "step": 34496, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 99, "lr": 0.03, "output_lr": 0.1, "step": 34848, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 100, "eval_accuracy": 0.1, "eval_loss": NaN, "eval_split": "validation", "lr": 0.03, "output_lr": 0.1, "step": 35200, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 101, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 35552, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 102, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 35904, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 103, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 36256, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 104, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 36608, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 105, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 36960, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 106, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 37312, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 107, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 37664, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 108, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 38016, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 109, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 38368, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 110, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 38720, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 111, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 39072, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 112, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 39424, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 113, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 39776, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 114, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 40128, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 115, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 40480, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 116, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 40832, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 117, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 41184, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 118, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 41536, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 119, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 41888, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 120, "eval_accuracy": 0.1, "eval_loss": NaN, "eval_split": "validation", "lr": 0.003, "output_lr": 0.010000000000000002, "step": 42240, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 121, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 42592, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 122, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 42944, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 123, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 43296, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 124, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 43648, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 125, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 44000, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 126, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 44352, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 127, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 44704, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 128, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 45056, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 129, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 45408, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 130, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 45760, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 131, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 46112, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 132, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 46464, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 133, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 46816, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 134, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 47168, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 135, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 47520, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 136, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 47872, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 137, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 48224, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 138, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 48576, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 139, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 48928, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 140, "eval_accuracy": 0.1, "eval_loss": NaN, "eval_split": "validation", "lr": 0.003, "output_lr": 0.010000000000000002, "step": 49280, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 141, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 49632, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 142, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 49984, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 143, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 50336, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 144, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 50688, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 145, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 51040, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 146, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 51392, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 147, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 51744, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 148, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 52096, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 149, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 52448, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 150, "lr": 0.003, "output_lr": 0.010000000000000002, "step": 52800, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 151, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 53152, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 152, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 53504, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 153, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 53856, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 154, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 54208, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 155, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 54560, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 156, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 54912, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 157, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 55264, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 158, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 55616, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 159, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 55968, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 160, "eval_accuracy": 0.1, "eval_loss": NaN, "eval_split": "validation", "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 56320, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 161, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 56672, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 162, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 57024, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 163, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 57376, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 164, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 57728, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 165, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 58080, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 166, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 58432, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 167, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 58784, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 168, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 59136, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 169, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 59488, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 170, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 59840, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 171, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 60192, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 172, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 60544, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 173, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 60896, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 174, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 61248, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 175, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 61600, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 176, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 61952, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 177, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 62304, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 178, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 62656, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 179, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 63008, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 180, "eval_accuracy": 0.1, "eval_loss": NaN, "eval_split": "validation", "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 63360, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 181, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 63712, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 182, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 64064, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 183, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 64416, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 184, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 64768, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 185, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 65120, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 186, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 65472, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 187, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 65824, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 188, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 66176, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 189, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 66528, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 190, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 66880, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 191, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 67232, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 192, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 67584, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 193, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 67936, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 194, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 68288, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 195, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 68640, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 196, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 68992, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 197, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 69344, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 198, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 69696, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 199, "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 70048, "train_examples": 45000, "train_loss": NaN }, { "calibration": { "calibration_mse": NaN, "prediction_target_cosine": NaN, "target_power": NaN }, "epoch": 200, "eval_accuracy": 0.1, "eval_loss": NaN, "eval_split": "validation", "lr": 0.00030000000000000003, "output_lr": 0.0010000000000000002, "step": 70400, "train_examples": 45000, "train_loss": NaN } ], "evaluation_protocol": { "test_evaluations": 0, "test_used_for_selection": false, "validation_evaluations": 11 }, "final": { "accuracy": 0.1, "epoch": 200, "evaluation_split": "validation", "finite": false, "loss": NaN, "step": 70400 }, "hardware": { "cuda_device_name": "NVIDIA GeForce GTX 1080", "cuda_visible_devices": "5", "device": "cuda:0", "device_total_memory_bytes": 8507949056, "peak_memory_allocated_bytes": 2164377600, "peak_memory_reserved_bytes": 3202351104, "torch_version": "2.3.1+cu118" }, "protocol_family": "oral_a_cifar_local_resnet_development", "provenance": { "git_commit": "a1d060e6d96d4ab973d16cd75149dd61899c7c95", "git_tracked_dirty": false }, "schema_version": 1, "split": { "cifar_source_files": [ { "bytes": 31035704, "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_1", "sha256": "54636561a3ce25bd3e19253c6b0d8538147b0ae398331ac4a2d86c6d987368cd" }, { "bytes": 31035320, "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_2", "sha256": "766b2cef9fbc745cf056b3152224f7cf77163b330ea9a15f9392beb8b89bc5a8" }, { "bytes": 31035999, "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_3", "sha256": "0f00d98ebfb30b3ec0ad19f9756dc2630b89003e10525f5e148445e82aa6a1f9" }, { "bytes": 31035696, "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_4", "sha256": "3f7bb240661948b8f4d53e36ec720d8306f5668bd0071dcb4e6c947f78e9682b" }, { "bytes": 31035623, "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/data_batch_5", "sha256": "d91802434d8376bbaeeadf58a737e3a1b12ac839077e931237e0dcd43adcb154" }, { "bytes": 31035526, "path": "/home/yurenh2/sdrn/data/cifar-10-batches-py/test_batch", "sha256": "f53d8d457504f7cff4ea9e021afcf0e0ad8e24a91f3fc42091b8adef61157831" } ], "dataset": "cifar10", "evaluation_split": "validation", "input_layout": "NCHW", "input_shape": [ 3, 32, 32 ], "loader_seed": 0, "normalization_mean": [ 0.49140000343322754, 0.4821999967098236, 0.4465000033378601 ], "normalization_std": [ 0.24699999392032623, 0.2434999942779541, 0.26159998774528503 ], "split_from_training_only": true, "split_seed": 2027, "test_examples": 10000, "train_examples": 45000, "training_augmentation": "random_crop_32_padding_4_zero_then_horizontal_flip_p0.5", "validation_class_counts": { "0": 500, "1": 500, "2": 500, "3": 500, "4": 500, "5": 500, "6": 500, "7": 500, "8": 500, "9": 500 }, "validation_examples": 5000, "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" }, "timing": { "apical_warmup_wall_s": 6.428055763244629, "evaluation_wall_s": 3.78330397605896, "predictor_warmup_wall_s": 0.0, "timing_excludes_data_loading_hashing_and_model_construction": true, "total_timed_wall_s": 4230.839186191559, "train_wall_s": 4220.594386100769 }, "work": { "apical_macs_per_example": 202176, "causal_scalar_observations": 35400, "components": { "apical_projection_macs": 1822171852800, "apical_regression_macs": 458049945600, "bp_reverse_macs_estimate": 0, "local_weight_correlation_macs": 364959360000000, "ordinary_forward_macs": 364959360000000, "perturbation_forward_macs": 183744872448000, "warmup_clean_forward_macs": 519053312000 }, "definition": "multiply-accumulates in conv/linear maps; one local weight correlation equals one forward-weight MAC count; BP reverse is estimated as one weight-gradient plus one activation-gradient convolution per forward convolution; elementwise nonlinearities and optimizer arithmetic excluded", "forward_macs_per_example": 40551040, "logical_batch_loss_queries": 35400, "per_example_cross_entropy_terms": 4531200, "total_clean_forward_examples": 9012800, "total_forward_equivalent_examples": 13544000, "total_macs_estimate": 916462867558400 } }