diff options
| author | YurenHao0426 <Blackhao0426@gmail.com> | 2026-07-23 08:13:02 -0500 |
|---|---|---|
| committer | YurenHao0426 <Blackhao0426@gmail.com> | 2026-07-23 08:13:02 -0500 |
| commit | 70e180c5ef5f78679f2e163ed3ee30873ee523bf (patch) | |
| tree | b3234a7a2311aafefd8e945c55e31551289d1058 | |
| parent | d289a46293e0b422c195746ad8b310d4b50ded23 (diff) | |
results: pass calibrated oral-B-v2 development gate
| -rw-r--r-- | results/bci_v2_calibrated_dev/bci_v2_calibrated_t26_m0.json | 461 | ||||
| -rw-r--r-- | results/bci_v2_calibrated_dev/bci_v2_calibrated_t27_m0.json | 461 | ||||
| -rw-r--r-- | results/bci_v2_calibrated_dev/bci_v2_calibrated_t28_m0.json | 461 | ||||
| -rw-r--r-- | results/bci_v2_calibrated_dev_gate.json | 148 |
4 files changed, 1531 insertions, 0 deletions
diff --git a/results/bci_v2_calibrated_dev/bci_v2_calibrated_t26_m0.json b/results/bci_v2_calibrated_dev/bci_v2_calibrated_t26_m0.json new file mode 100644 index 0000000..b2fc4f3 --- /dev/null +++ b/results/bci_v2_calibrated_dev/bci_v2_calibrated_t26_m0.json @@ -0,0 +1,461 @@ +{ + "args": { + "model_seed": 0, + "outdir": "results/bci_v2_calibrated_dev", + "task_seed": 26 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12882, + "acute_outcome_lesion": 12882, + "intact": 12882 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.9664530754089355, + "median": 1.9004390239715576, + "minimum": 1.6664398908615112 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 550026, + "targets": [ + 1.859693741798401, + 1.8864485919475555, + 1.9004749059677124, + 1.9156511783599854, + 1.929451322555542 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.859693741798401, + 1.8864485919475555, + 1.9004749059677124, + 1.9156511783599854, + 1.929451322555542 + ], + "trajectory_seeds": { + "1": 586000, + "2": 586001, + "3": 586002, + "4": 586003, + "5": 586004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 540026 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9942126274108887, + "training_wall_s": 0.21116646006703377 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06177661940455437, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21275918185710907 + }, + "intact": { + "cost": { + "active_state_episode_steps": 16204, + "cursor_scalar_observations": 8554, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4277, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5207661390304565, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.203125, + 0.6875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 263, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9405146241188049, + "training_wall_s": 0.2710344232618809 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 15936, + "cursor_scalar_observations": 8436, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4218, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.45849862694740295, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.265625, + 0.796875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 267, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.2009308896958828 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 19102, + "cursor_scalar_observations": 9788, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4894, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.18997597694396973, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.15625, + 0.25, + 0.671875, + 0.8125, + 0.96875, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 641, + "final_success": 1.0, + "late_success": 0.9895833333333334, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 0.9652196764945984, + "training_wall_s": 0.22571271285414696 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06284985691308975, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9942126870155334, + "training_wall_s": 0.20297767594456673 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 962.34375, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_seeds_touched": false, + "failed_target_gate_sha256": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "hyperparameter_selection": false, + "name": "oral_b_v2_calibrated_recovery_development_v1", + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "selection_split": "development_validation" + }, + "provenance": { + "git_commit": "d289a46293e0b422c195746ad8b310d4b50ded23", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5002197802197802, + "acute_outcome_lesion_role_aligned_separation": 0.0005471409294360297, + "causal_role_sign_inversion_index": 0.03878189930912869, + "challenge_episodes": 640, + "challenge_failure_count": 315, + "challenge_success_count": 325, + "challenge_success_fraction": 0.5078125, + "critic_contribution_value_prediction_corr": 0.9999999999995302, + "decoder_distance_residual_corr": 0.11469715057185319, + "mean_abs_raw_soma_corr": 0.9992015754075411, + "mean_abs_residual_soma_corr": 0.0563586791161443, + "mean_critic_expectedness_contribution": 0.24124016763074801, + "nonterminal_training_events": 15308, + "raw_minus_residual_abs_soma_corr": 0.9428428962913967, + "role_aligned_error_cv_corr": 0.32966432771863374, + "role_aligned_velocity_cv_corr": 0.9988614599944285, + "surrounding_event_decoder_balanced_acc": 0.5539961311201976, + "target_success_fraction": { + "1.859693741798401": 0.828125, + "1.8864485919475555": 0.71875, + "1.9004749059677124": 0.4453125, + "1.9156511783599854": 0.328125, + "1.929451322555542": 0.21875 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.3956960119743165, + "terminal_previous_soma_outcome_balanced_acc": 0.7272771672771673, + "terminal_residual_minus_previous_soma_acc": 0.26796092796092796, + "terminal_residual_outcome_balanced_acc": 0.9952380952380953, + "terminal_role_aligned_outcome_separation": 0.3962431529037525, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6691971322757948 + }, + "split": "development_validation", + "wall_s": 2.583215858787298, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602600, + "role_cosine_after_warmup": 0.9944602251052856, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602600, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602600, + "role_cosine_after_warmup": 0.9944602251052856, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602600, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602600, + "role_cosine_after_warmup": 0.9944602251052856, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602600, + "role_cosine_after_warmup": 0.9944602251052856, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_dev/bci_v2_calibrated_t27_m0.json b/results/bci_v2_calibrated_dev/bci_v2_calibrated_t27_m0.json new file mode 100644 index 0000000..f1e11a3 --- /dev/null +++ b/results/bci_v2_calibrated_dev/bci_v2_calibrated_t27_m0.json @@ -0,0 +1,461 @@ +{ + "args": { + "model_seed": 0, + "outdir": "results/bci_v2_calibrated_dev", + "task_seed": 27 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13244, + "acute_outcome_lesion": 13244, + "intact": 13244 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8732032775878906, + "median": 1.749955177307129, + "minimum": 1.5971075296401978 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 550027, + "targets": [ + 1.7139501571655273, + 1.7322039246559142, + 1.750077724456787, + 1.7649245738983155, + 1.783210277557373 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7139501571655273, + 1.7322039246559142, + 1.750077724456787, + 1.7649245738983155, + 1.783210277557373 + ], + "trajectory_seeds": { + "1": 587000, + "2": 587001, + "3": 587002, + "4": 587003, + "5": 587004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 540027 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 14474, + "cursor_scalar_observations": 7798, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3899, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.015625, + 0.1875, + 0.609375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9887208938598633, + "training_wall_s": 0.16654988750815392 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06176371872425079, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21138394996523857 + }, + "intact": { + "cost": { + "active_state_episode_steps": 11369, + "cursor_scalar_observations": 6412, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3206, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.508084774017334, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.109375, + 0.671875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9776318073272705, + "training_wall_s": 0.24492322281002998 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 11256, + "cursor_scalar_observations": 6362, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3181, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.520680844783783, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.125, + 0.6875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.16702108830213547 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 17895, + "cursor_scalar_observations": 9242, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4621, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.22657130658626556, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.078125, + 0.265625, + 0.390625, + 0.71875, + 0.859375, + 0.875, + 0.96875, + 0.9375, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 1375, + "final_success": 0.984375, + "late_success": 0.96875, + "learning_gain": 0.96875, + "role_cosine_after_training": 0.9808546304702759, + "training_wall_s": 0.2095155566930771 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06274955719709396, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9958968758583069, + "training_wall_s": 0.18501129373908043 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 947.375, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_seeds_touched": false, + "failed_target_gate_sha256": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "hyperparameter_selection": false, + "name": "oral_b_v2_calibrated_recovery_development_v1", + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "selection_split": "development_validation" + }, + "provenance": { + "git_commit": "d289a46293e0b422c195746ad8b310d4b50ded23", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5622654784240151, + "acute_outcome_lesion_role_aligned_separation": -0.004976110356813113, + "causal_role_sign_inversion_index": 0.03865690938282479, + "challenge_episodes": 640, + "challenge_failure_count": 328, + "challenge_success_count": 312, + "challenge_success_fraction": 0.4875, + "critic_contribution_value_prediction_corr": 0.9999999999991853, + "decoder_distance_residual_corr": 0.08504896381928725, + "mean_abs_raw_soma_corr": 0.9990959548954148, + "mean_abs_residual_soma_corr": 0.05868808106050728, + "mean_critic_expectedness_contribution": 0.35806743977971284, + "nonterminal_training_events": 10473, + "raw_minus_residual_abs_soma_corr": 0.9404078738349075, + "role_aligned_error_cv_corr": 0.39857193059373297, + "role_aligned_velocity_cv_corr": 0.9978555331143931, + "surrounding_event_decoder_balanced_acc": 0.541416057442577, + "target_success_fraction": { + "1.7139501571655273": 0.796875, + "1.7322039246559142": 0.6640625, + "1.750077724456787": 0.390625, + "1.7649245738983155": 0.3828125, + "1.783210277557373": 0.203125 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.40359597715821127, + "terminal_previous_soma_outcome_balanced_acc": 0.7347951844903065, + "terminal_residual_minus_previous_soma_acc": 0.26520481550969355, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.39861986680139816, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.5992836025206602 + }, + "split": "development_validation", + "wall_s": 2.259972095489502, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602700, + "role_cosine_after_warmup": 0.9947453737258911, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602700, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602700, + "role_cosine_after_warmup": 0.9947453737258911, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602700, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602700, + "role_cosine_after_warmup": 0.9947453737258911, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602700, + "role_cosine_after_warmup": 0.9947453737258911, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_dev/bci_v2_calibrated_t28_m0.json b/results/bci_v2_calibrated_dev/bci_v2_calibrated_t28_m0.json new file mode 100644 index 0000000..3880956 --- /dev/null +++ b/results/bci_v2_calibrated_dev/bci_v2_calibrated_t28_m0.json @@ -0,0 +1,461 @@ +{ + "args": { + "model_seed": 0, + "outdir": "results/bci_v2_calibrated_dev", + "task_seed": 28 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12957, + "acute_outcome_lesion": 12957, + "intact": 12957 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.9806733131408691, + "median": 1.9220820665359497, + "minimum": 1.6861028671264648 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 550028, + "targets": [ + 1.8872586727142333, + 1.9072148621082305, + 1.9221965074539185, + 1.9328129887580872, + 1.9453917741775513 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.8872586727142333, + 1.9072148621082305, + 1.9221965074539185, + 1.9328129887580872, + 1.9453917741775513 + ], + "trajectory_seeds": { + "1": 588000, + "2": 588001, + "3": 588002, + "4": 588003, + "5": 588004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 540028 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 17384, + "cursor_scalar_observations": 9062, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4531, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.015625, + 0.0625, + 0.296875, + 0.859375, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 261, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9773973226547241, + "training_wall_s": 0.1755989007651806 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06155245006084442, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21381374821066856 + }, + "intact": { + "cost": { + "active_state_episode_steps": 14224, + "cursor_scalar_observations": 7682, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3841, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.533806562423706, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.03125, + 0.203125, + 0.890625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 260, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9761794805526733, + "training_wall_s": 0.258306160569191 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 14083, + "cursor_scalar_observations": 7614, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3807, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.4723714292049408, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.046875, + 0.0625, + 0.234375, + 0.890625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.18617762252688408 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 18917, + "cursor_scalar_observations": 9704, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4852, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.17078045010566711, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.03125, + 0.078125, + 0.265625, + 0.5, + 0.78125, + 0.84375, + 0.953125, + 0.9375, + 0.984375 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 1684, + "final_success": 0.96875, + "late_success": 0.9583333333333334, + "learning_gain": 0.9583333333333334, + "role_cosine_after_training": 0.9445701241493225, + "training_wall_s": 0.21972168236970901 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06252607703208923, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9957386255264282, + "training_wall_s": 0.1857893168926239 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 954.734375, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_seeds_touched": false, + "failed_target_gate_sha256": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "hyperparameter_selection": false, + "name": "oral_b_v2_calibrated_recovery_development_v1", + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "selection_split": "development_validation" + }, + "provenance": { + "git_commit": "d289a46293e0b422c195746ad8b310d4b50ded23", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.540903540903541, + "acute_outcome_lesion_role_aligned_separation": 0.010854148777368705, + "causal_role_sign_inversion_index": 0.03921735111033796, + "challenge_episodes": 640, + "challenge_failure_count": 315, + "challenge_success_count": 325, + "challenge_success_fraction": 0.5078125, + "critic_contribution_value_prediction_corr": 0.9999999999994963, + "decoder_distance_residual_corr": 0.13642571216406077, + "mean_abs_raw_soma_corr": 0.9990992136226702, + "mean_abs_residual_soma_corr": 0.05195367815050812, + "mean_critic_expectedness_contribution": 0.24424682945902534, + "nonterminal_training_events": 13328, + "raw_minus_residual_abs_soma_corr": 0.9471455354721621, + "role_aligned_error_cv_corr": 0.34897921307347385, + "role_aligned_velocity_cv_corr": 0.9982339724621896, + "surrounding_event_decoder_balanced_acc": 0.5528030824638526, + "target_success_fraction": { + "1.8872586727142333": 0.78125, + "1.9072148621082305": 0.640625, + "1.9221965074539185": 0.4765625, + "1.9328129887580872": 0.40625, + "1.9453917741775513": 0.234375 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.38343148953250883, + "terminal_previous_soma_outcome_balanced_acc": 0.7603663003663004, + "terminal_residual_minus_previous_soma_acc": 0.23804639804639804, + "terminal_residual_outcome_balanced_acc": 0.9984126984126984, + "terminal_role_aligned_outcome_separation": 0.39428563830987756, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6492547593887157 + }, + "split": "development_validation", + "wall_s": 2.422760460525751, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602800, + "role_cosine_after_warmup": 0.9957716464996338, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602800, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602800, + "role_cosine_after_warmup": 0.9957716464996338, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602800, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602800, + "role_cosine_after_warmup": 0.9957716464996338, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602800, + "role_cosine_after_warmup": 0.9957716464996338, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_dev_gate.json b/results/bci_v2_calibrated_dev_gate.json new file mode 100644 index 0000000..515f0e6 --- /dev/null +++ b/results/bci_v2_calibrated_dev_gate.json @@ -0,0 +1,148 @@ +{ + "calibrated_confirmation_opened": true, + "calibrated_targets_by_task_seed": { + "26": [ + 1.859693741798401, + 1.8864485919475555, + 1.9004749059677124, + 1.9156511783599854, + 1.929451322555542 + ], + "27": [ + 1.7139501571655273, + 1.7322039246559142, + 1.750077724456787, + 1.7649245738983155, + 1.783210277557373 + ], + "28": [ + 1.8872586727142333, + 1.9072148621082305, + 1.9221965074539185, + 1.9328129887580872, + 1.9453917741775513 + ] + }, + "calibration_quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "challenge_success_fraction_by_task_seed": [ + 0.5078125, + 0.4875, + 0.5078125 + ], + "checks_by_task_seed": { + "26": { + "challenge_fraction_between_0p15_and_0p85": true, + "critic_contribution_value_corr_at_least_0p95": true, + "critic_expectedness_at_least_0p05": true, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": true, + "fixed_role_gap_at_least_0p20": true, + "learning_gain_at_least_0p10": true, + "oracle_deficit_at_most_0p10": true, + "outcome_lesion_separation_drop_at_least_0p20": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "role_cosine_at_least_0p80": true, + "sign_inversion_at_least_0p01": true, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p75": true, + "terminal_role_separation_at_least_0p20": true, + "velocity_advantage_at_least_0p05": true + }, + "27": { + "challenge_fraction_between_0p15_and_0p85": true, + "critic_contribution_value_corr_at_least_0p95": true, + "critic_expectedness_at_least_0p05": true, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": true, + "fixed_role_gap_at_least_0p20": true, + "learning_gain_at_least_0p10": true, + "oracle_deficit_at_most_0p10": true, + "outcome_lesion_separation_drop_at_least_0p20": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "role_cosine_at_least_0p80": true, + "sign_inversion_at_least_0p01": true, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p75": true, + "terminal_role_separation_at_least_0p20": true, + "velocity_advantage_at_least_0p05": true + }, + "28": { + "challenge_fraction_between_0p15_and_0p85": true, + "critic_contribution_value_corr_at_least_0p95": true, + "critic_expectedness_at_least_0p05": true, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": true, + "fixed_role_gap_at_least_0p20": true, + "learning_gain_at_least_0p10": true, + "oracle_deficit_at_most_0p10": true, + "outcome_lesion_separation_drop_at_least_0p20": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "role_cosine_at_least_0p80": true, + "sign_inversion_at_least_0p01": true, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p75": true, + "terminal_role_separation_at_least_0p20": true, + "velocity_advantage_at_least_0p05": true + } + }, + "complete_grid": true, + "confirmation_seeds_touched": false, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "intact_final_by_task_seed": [ + 1.0, + 1.0, + 1.0 + ], + "protocol": "oral_b_v2_calibrated_recovery_development_v1", + "review_score_after": 7, + "review_score_before": 7, + "score_change_rule": "development validation never changes the formal score", + "source_commit": "d289a46293e0b422c195746ad8b310d4b50ded23", + "source_sha256": { + "results/bci_v2_calibrated_dev/bci_v2_calibrated_t26_m0.json": "3ad58c0d4ee6aef1835f281c60c63b7c309ff23971692ba97631fd81d43c7165", + "results/bci_v2_calibrated_dev/bci_v2_calibrated_t27_m0.json": "3c61254dbf22ba1cb550f4c35aecba2375c754ac64ad383c642ecaaff8085670", + "results/bci_v2_calibrated_dev/bci_v2_calibrated_t28_m0.json": "b09933f8776ae1941adb0db543d1f5df91e9287cb11b2536e27e3b9caafc5812" + }, + "status": "passed", + "terminal_outcome_accuracy_by_task_seed": [ + 0.9952380952380953, + 1.0, + 0.9984126984126984 + ] +} |
