From da109afa23146988188c3af910c6a6205f0c63d9 Mon Sep 17 00:00:00 2001 From: YurenHao0426 Date: Wed, 22 Jul 2026 12:46:08 -0500 Subject: analysis: localize early-layer feedback bottlenecks --- results/oral_a_representation_diagnosis.json | 273 +++++++++++++++++++++++++++ 1 file changed, 273 insertions(+) create mode 100644 results/oral_a_representation_diagnosis.json (limited to 'results/oral_a_representation_diagnosis.json') diff --git a/results/oral_a_representation_diagnosis.json b/results/oral_a_representation_diagnosis.json new file mode 100644 index 0000000..3c170c1 --- /dev/null +++ b/results/oral_a_representation_diagnosis.json @@ -0,0 +1,273 @@ +{ + "cosine": { + "channel_gated_cv": { + "all_layer_mean": 0.26860839708781414, + "early_third_mean": 0.02396580002561188, + "per_layer": [ + 0.011761176795650953, + 0.01509529795660324, + 0.02110834681304405, + 0.02648881734378023, + 0.035379675557930944, + 0.03396148568666185, + 0.04737202878125393, + 0.08733494968419549, + 0.11412488007354467, + 0.11111145838920782, + 0.16484829973665457, + 0.1705562681820029, + 0.22815721287819002, + 0.401197490658692, + 0.5625015542517685, + 0.5668841474011743, + 0.7651559832644279, + 0.7405204712136877, + 0.999999999999998 + ] + }, + "local_context_cv": { + "all_layer_mean": 0.2657763621141079, + "early_third_mean": 0.022457713873793972, + "per_layer": [ + 0.011563062975842375, + 0.013967016700256895, + 0.019497958665444695, + 0.025315213917818324, + 0.03192140408252579, + 0.032481626900875755, + 0.04422040389025903, + 0.08206046492345852, + 0.11003918030189887, + 0.10851704119726613, + 0.1613744067292345, + 0.167258052602825, + 0.22673659297189666, + 0.39712579796193304, + 0.5577716450313811, + 0.5599644189104344, + 0.7632643600906147, + 0.7366722323140862, + 0.9999999999999978 + ] + }, + "per_example_spatial_oracle": { + "all_layer_mean": 0.3080300858793031, + "early_third_mean": 0.05494018487684554, + "per_layer": [ + 0.044284993661542316, + 0.050289163913581, + 0.054174960251254504, + 0.056859289797845186, + 0.06385232305504676, + 0.06018037858180342, + 0.07359532477033978, + 0.1412692499078675, + 0.16052161434051526, + 0.15835344551911612, + 0.20160746652301614, + 0.20989211288493392, + 0.2581259209822714, + 0.4772294334388024, + 0.6118273095794977, + 0.6367033659113905, + 0.7892394590882331, + 0.8045658194996996, + 1.0000000000000022 + ] + }, + "spatial_template_cv": { + "all_layer_mean": 0.2988933125374179, + "early_third_mean": 0.04796723142241297, + "per_layer": [ + 0.018794834494927203, + 0.028104944902399245, + 0.03589688012133678, + 0.04943569370504626, + 0.06487004742738658, + 0.09070098788338177, + 0.10403720218754312, + 0.12806722415370214, + 0.15473667864101381, + 0.2100427890594658, + 0.251471876051357, + 0.29367024000272923, + 0.3516688770876262, + 0.36271913833514663, + 0.48618596240775314, + 0.5601553963417819, + 0.7149574547592554, + 0.77345671064909, + 0.9999999999999973 + ] + } + }, + "interpretation": { + "channel_gated_cv": "actual v2 vectorizer family fit on 32 examples and evaluated on 32 disjoint examples", + "local_context_cv": "diagnostic four-field family adding local spatial average and cross-channel somatic context", + "per_example_spatial_oracle": "upper bound for the two [1,tanh(h)] fields with unconstrained per-example/channel coefficients", + "spatial_template_cv": "output-error linear map with an independent coefficient at every hidden unit", + "status": "post_failure_diagnosis_not_a_learning_result_or_gate" + }, + "network": { + "depth": 20, + "forward_state": "deterministic_initialization_no_updates", + "normalization": "batchnorm", + "residual_scale": 1.0, + "seed": 0, + "width": 16 + }, + "prediction_energy_over_target_energy": { + "channel_gated_cv": { + "all_layer_mean": 0.18153600823060526, + "early_third_mean": 0.002141493001862695, + "per_layer": [ + 0.0013493838251117957, + 0.0018821866339346622, + 0.001696421773313066, + 0.001957989772753797, + 0.002779034484324995, + 0.003183941521737851, + 0.004221306553680124, + 0.016149756754974574, + 0.021920567376265642, + 0.02088059044942729, + 0.0365591553150065, + 0.0398603168872378, + 0.0655479631539235, + 0.21288315903697133, + 0.36735600776651145, + 0.39189274768920795, + 0.6184532017563165, + 0.6406104266618678, + 0.9999999989689331 + ] + }, + "local_context_cv": { + "all_layer_mean": 0.19307878388438796, + "early_third_mean": 0.0034804679550285975, + "per_layer": [ + 0.002811484345442635, + 0.0030526487624646416, + 0.0029164019568946173, + 0.0033475132507695907, + 0.004236770387682787, + 0.004517989026917315, + 0.005602177120465608, + 0.020577920086456147, + 0.026298887646854674, + 0.026779671005441024, + 0.04293551458003673, + 0.046946976827938976, + 0.07304993437673593, + 0.25110846414727345, + 0.39983404016800456, + 0.4403499134572238, + 0.6400042870848195, + 0.6741263003721254, + 0.999999999199824 + ] + }, + "per_example_spatial_oracle": { + "all_layer_mean": 0.18659338577180437, + "early_third_mean": 0.0030446135866956468, + "per_layer": [ + 0.0019590616394833775, + 0.0024993103774871434, + 0.002925141382872601, + 0.0031998702722740354, + 0.004054197420553255, + 0.0036301004275034673, + 0.005254602524406208, + 0.019885848610928476, + 0.025921735678432582, + 0.024952912936957894, + 0.04109786003346419, + 0.04484580079931531, + 0.06838933561711134, + 0.2341008355593867, + 0.3779443971902448, + 0.4102824117515338, + 0.6264783686368899, + 0.6478525388054378, + 1.0 + ] + }, + "spatial_template_cv": { + "all_layer_mean": 0.578876932893507, + "early_third_mean": 0.457647070414877, + "per_layer": [ + 0.4541521195275675, + 0.47896384166837286, + 0.448121193317785, + 0.44093294951602213, + 0.45548109382727925, + 0.4682312246322352, + 0.47915529592237, + 0.48488773706669025, + 0.49772630973811416, + 0.5115627889679282, + 0.5303103557041301, + 0.556252488968008, + 0.5816302553670893, + 0.6011432068873668, + 0.6639513115023615, + 0.7067060521061842, + 0.7903486506759895, + 0.8491048527148057, + 0.9999999968663323 + ] + } + }, + "probe": { + "batchnorm_cv": "fit and evaluation halves use separate 32-example batch statistics and exact-gradient graphs", + "evaluation_examples": 32, + "examples": 64, + "fit_examples": 32, + "source": "unaugmented first training-prefix examples", + "test_examples_touched": 0 + }, + "protocol": "oral_a_post_failure_representation_diagnosis_v1", + "provenance": { + "git_commit": "ba50c8fc286f6071c688a12606216cfb871a7ee0", + "git_tracked_dirty": false + }, + "split": { + "dataset": "cifar10", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "none", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + } +} -- cgit v1.2.3