diff options
| author | YurenHao0426 <Blackhao0426@gmail.com> | 2026-07-22 12:46:08 -0500 |
|---|---|---|
| committer | YurenHao0426 <Blackhao0426@gmail.com> | 2026-07-22 12:46:08 -0500 |
| commit | da109afa23146988188c3af910c6a6205f0c63d9 (patch) | |
| tree | 3bd096b1be9ad91a0be6ab0dc037568473cb0345 /results | |
| parent | ba50c8fc286f6071c688a12606216cfb871a7ee0 (diff) | |
analysis: localize early-layer feedback bottlenecks
Diffstat (limited to 'results')
| -rw-r--r-- | results/oral_a_failure_diagnosis.json | 14 | ||||
| -rw-r--r-- | results/oral_a_representation_diagnosis.json | 273 |
2 files changed, 285 insertions, 2 deletions
diff --git a/results/oral_a_failure_diagnosis.json b/results/oral_a_failure_diagnosis.json index 5239e8b..1b142df 100644 --- a/results/oral_a_failure_diagnosis.json +++ b/results/oral_a_failure_diagnosis.json @@ -44,14 +44,17 @@ "target_power_growth_epoch1_to_87": 2126827049.3932126 }, "diagnosis": { - "classification": "unit_target_regression_snr_inadequate_before_runaway", + "classification": "early_credit_limited_by_estimator_efficiency_and_feedback_capacity", "evidence": [ "instantaneous prediction-target cosine remains effectively zero", "calibration MSE remains indistinguishable from stochastic target power", "target power grows by orders of magnitude before nonfiniteness", - "matched frozen-forward unit targets later yield only 0.0011 early-layer alignment" + "matched frozen-forward unit targets later yield only 0.0011 early-layer alignment", + "structured learning reaches 0.0072 versus a 0.0240 cross-validated family oracle", + "unconstrained coefficients raise the same spatial basis oracle only to 0.0549" ], "interpretation_limit": "instantaneous target metrics alone do not prove zero conditional learning; structured v2 has low instantaneous cosine yet positive exact-gradient alignment", + "next_design_constraint": "do not add the tested naive local-average/channel-mean bases; improve causal sample efficiency and use a more informative hierarchical or high-level contextual feedback signal", "next_test_outcome": "representable-subspace perturbation improved exact alignment but failed the frozen early-layer causal-capture gate" }, "frozen_outcome": { @@ -72,6 +75,12 @@ "loss": null } }, + "post_failure_representation_refinement": { + "channel_gated_cv_oracle_early_third_alignment": 0.02396580002561188, + "naive_local_context_cv_early_third_alignment": 0.022457713873793972, + "per_example_spatial_oracle_early_third_alignment": 0.05494018487684554, + "trained_structured_early_third_alignment": 0.007209055746595065 + }, "post_failure_v2_refinement": { "structured_all_layer_alignment": 0.05273995646520665, "structured_early_third_alignment": 0.007209055746595065, @@ -93,6 +102,7 @@ "source_paths": { "full": "results/oral_a_dev/sdil_full_r20_s0.json", "gate": "results/oral_a_full_gate.json", + "representation": "results/oral_a_representation_diagnosis.json", "short": "results/oral_a_short/sdil_channel_gated_lr0.03.json", "v2_gate": "results/oral_a_v2_calibration_gate.json" } diff --git a/results/oral_a_representation_diagnosis.json b/results/oral_a_representation_diagnosis.json new file mode 100644 index 0000000..3c170c1 --- /dev/null +++ b/results/oral_a_representation_diagnosis.json @@ -0,0 +1,273 @@ +{ + "cosine": { + "channel_gated_cv": { + "all_layer_mean": 0.26860839708781414, + "early_third_mean": 0.02396580002561188, + "per_layer": [ + 0.011761176795650953, + 0.01509529795660324, + 0.02110834681304405, + 0.02648881734378023, + 0.035379675557930944, + 0.03396148568666185, + 0.04737202878125393, + 0.08733494968419549, + 0.11412488007354467, + 0.11111145838920782, + 0.16484829973665457, + 0.1705562681820029, + 0.22815721287819002, + 0.401197490658692, + 0.5625015542517685, + 0.5668841474011743, + 0.7651559832644279, + 0.7405204712136877, + 0.999999999999998 + ] + }, + "local_context_cv": { + "all_layer_mean": 0.2657763621141079, + "early_third_mean": 0.022457713873793972, + "per_layer": [ + 0.011563062975842375, + 0.013967016700256895, + 0.019497958665444695, + 0.025315213917818324, + 0.03192140408252579, + 0.032481626900875755, + 0.04422040389025903, + 0.08206046492345852, + 0.11003918030189887, + 0.10851704119726613, + 0.1613744067292345, + 0.167258052602825, + 0.22673659297189666, + 0.39712579796193304, + 0.5577716450313811, + 0.5599644189104344, + 0.7632643600906147, + 0.7366722323140862, + 0.9999999999999978 + ] + }, + "per_example_spatial_oracle": { + "all_layer_mean": 0.3080300858793031, + "early_third_mean": 0.05494018487684554, + "per_layer": [ + 0.044284993661542316, + 0.050289163913581, + 0.054174960251254504, + 0.056859289797845186, + 0.06385232305504676, + 0.06018037858180342, + 0.07359532477033978, + 0.1412692499078675, + 0.16052161434051526, + 0.15835344551911612, + 0.20160746652301614, + 0.20989211288493392, + 0.2581259209822714, + 0.4772294334388024, + 0.6118273095794977, + 0.6367033659113905, + 0.7892394590882331, + 0.8045658194996996, + 1.0000000000000022 + ] + }, + "spatial_template_cv": { + "all_layer_mean": 0.2988933125374179, + "early_third_mean": 0.04796723142241297, + "per_layer": [ + 0.018794834494927203, + 0.028104944902399245, + 0.03589688012133678, + 0.04943569370504626, + 0.06487004742738658, + 0.09070098788338177, + 0.10403720218754312, + 0.12806722415370214, + 0.15473667864101381, + 0.2100427890594658, + 0.251471876051357, + 0.29367024000272923, + 0.3516688770876262, + 0.36271913833514663, + 0.48618596240775314, + 0.5601553963417819, + 0.7149574547592554, + 0.77345671064909, + 0.9999999999999973 + ] + } + }, + "interpretation": { + "channel_gated_cv": "actual v2 vectorizer family fit on 32 examples and evaluated on 32 disjoint examples", + "local_context_cv": "diagnostic four-field family adding local spatial average and cross-channel somatic context", + "per_example_spatial_oracle": "upper bound for the two [1,tanh(h)] fields with unconstrained per-example/channel coefficients", + "spatial_template_cv": "output-error linear map with an independent coefficient at every hidden unit", + "status": "post_failure_diagnosis_not_a_learning_result_or_gate" + }, + "network": { + "depth": 20, + "forward_state": "deterministic_initialization_no_updates", + "normalization": "batchnorm", + "residual_scale": 1.0, + "seed": 0, + "width": 16 + }, + "prediction_energy_over_target_energy": { + "channel_gated_cv": { + "all_layer_mean": 0.18153600823060526, + "early_third_mean": 0.002141493001862695, + "per_layer": [ + 0.0013493838251117957, + 0.0018821866339346622, + 0.001696421773313066, + 0.001957989772753797, + 0.002779034484324995, + 0.003183941521737851, + 0.004221306553680124, + 0.016149756754974574, + 0.021920567376265642, + 0.02088059044942729, + 0.0365591553150065, + 0.0398603168872378, + 0.0655479631539235, + 0.21288315903697133, + 0.36735600776651145, + 0.39189274768920795, + 0.6184532017563165, + 0.6406104266618678, + 0.9999999989689331 + ] + }, + "local_context_cv": { + "all_layer_mean": 0.19307878388438796, + "early_third_mean": 0.0034804679550285975, + "per_layer": [ + 0.002811484345442635, + 0.0030526487624646416, + 0.0029164019568946173, + 0.0033475132507695907, + 0.004236770387682787, + 0.004517989026917315, + 0.005602177120465608, + 0.020577920086456147, + 0.026298887646854674, + 0.026779671005441024, + 0.04293551458003673, + 0.046946976827938976, + 0.07304993437673593, + 0.25110846414727345, + 0.39983404016800456, + 0.4403499134572238, + 0.6400042870848195, + 0.6741263003721254, + 0.999999999199824 + ] + }, + "per_example_spatial_oracle": { + "all_layer_mean": 0.18659338577180437, + "early_third_mean": 0.0030446135866956468, + "per_layer": [ + 0.0019590616394833775, + 0.0024993103774871434, + 0.002925141382872601, + 0.0031998702722740354, + 0.004054197420553255, + 0.0036301004275034673, + 0.005254602524406208, + 0.019885848610928476, + 0.025921735678432582, + 0.024952912936957894, + 0.04109786003346419, + 0.04484580079931531, + 0.06838933561711134, + 0.2341008355593867, + 0.3779443971902448, + 0.4102824117515338, + 0.6264783686368899, + 0.6478525388054378, + 1.0 + ] + }, + "spatial_template_cv": { + "all_layer_mean": 0.578876932893507, + "early_third_mean": 0.457647070414877, + "per_layer": [ + 0.4541521195275675, + 0.47896384166837286, + 0.448121193317785, + 0.44093294951602213, + 0.45548109382727925, + 0.4682312246322352, + 0.47915529592237, + 0.48488773706669025, + 0.49772630973811416, + 0.5115627889679282, + 0.5303103557041301, + 0.556252488968008, + 0.5816302553670893, + 0.6011432068873668, + 0.6639513115023615, + 0.7067060521061842, + 0.7903486506759895, + 0.8491048527148057, + 0.9999999968663323 + ] + } + }, + "probe": { + "batchnorm_cv": "fit and evaluation halves use separate 32-example batch statistics and exact-gradient graphs", + "evaluation_examples": 32, + "examples": 64, + "fit_examples": 32, + "source": "unaugmented first training-prefix examples", + "test_examples_touched": 0 + }, + "protocol": "oral_a_post_failure_representation_diagnosis_v1", + "provenance": { + "git_commit": "ba50c8fc286f6071c688a12606216cfb871a7ee0", + "git_tracked_dirty": false + }, + "split": { + "dataset": "cifar10", + "input_layout": "NCHW", + "input_shape": [ + 3, + 32, + 32 + ], + "loader_seed": 0, + "normalization_mean": [ + 0.49140000343322754, + 0.4821999967098236, + 0.4465000033378601 + ], + "normalization_std": [ + 0.24699999392032623, + 0.2434999942779541, + 0.26159998774528503 + ], + "split_from_training_only": true, + "split_seed": 2027, + "test_examples": 10000, + "train_examples": 10000, + "training_augmentation": "none", + "validation_class_counts": { + "0": 500, + "1": 500, + "2": 500, + "3": 500, + "4": 500, + "5": 500, + "6": 500, + "7": 500, + "8": 500, + "9": 500 + }, + "validation_examples": 5000, + "validation_index_sha256": "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b" + } +} |
