diff options
| author | YurenHao0426 <Blackhao0426@gmail.com> | 2026-07-23 08:15:34 -0500 |
|---|---|---|
| committer | YurenHao0426 <Blackhao0426@gmail.com> | 2026-07-23 08:15:34 -0500 |
| commit | 9a8c0570f8d96cc3bca200ed872323e44e60f55f (patch) | |
| tree | dfb6ffd82a73732344d3527df404dec76564b46c | |
| parent | 70e180c5ef5f78679f2e163ed3ee30873ee523bf (diff) | |
results: pass untouched calibrated oral-B-v2 confirmation
31 files changed, 14425 insertions, 0 deletions
diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m0.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m0.json new file mode 100644 index 0000000..fa05fb7 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m0.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 0, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 30 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13178, + "acute_outcome_lesion": 13178, + "intact": 13178 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8811628818511963, + "median": 1.746944546699524, + "minimum": 1.5939404964447021 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600030, + "targets": [ + 1.7077688932418824, + 1.7300113141536713, + 1.74697607755661, + 1.7630531013011932, + 1.7839327335357666 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7077688932418824, + 1.7300113141536713, + 1.74697607755661, + 1.7630531013011932, + 1.7839327335357666 + ], + "trajectory_seeds": { + "1": 640000, + "2": 640001, + "3": 640002, + "4": 640003, + "5": 640004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590030 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 18767, + "cursor_scalar_observations": 9726, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4863, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.015625, + 0.203125, + 0.3125, + 0.984375, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 257, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9901195764541626, + "training_wall_s": 0.19714121520519257 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06020602583885193, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.22645844146609306 + }, + "intact": { + "cost": { + "active_state_episode_steps": 13394, + "cursor_scalar_observations": 7302, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3651, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.48879820108413696, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.015625, + 0.0625, + 0.515625, + 0.984375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 259, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9855557680130005, + "training_wall_s": 0.25578784570097923 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 12939, + "cursor_scalar_observations": 7110, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3555, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5247551798820496, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.015625, + 0.125, + 0.71875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 257, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.19350211694836617 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 18508, + "cursor_scalar_observations": 9538, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4769, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.20170171558856964, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.015625, + 0.015625, + 0.15625, + 0.3125, + 0.515625, + 0.796875, + 0.890625, + 0.96875, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 798, + "final_success": 1.0, + "late_success": 0.9895833333333334, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 0.9825987219810486, + "training_wall_s": 0.22674556821584702 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.061336178332567215, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9950096011161804, + "training_wall_s": 0.20278817787766457 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 954.86328125, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 0, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 30 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.6241805160672588, + "acute_outcome_lesion_role_aligned_separation": -0.001996671326479016, + "causal_role_sign_inversion_index": 0.038511405209982176, + "challenge_episodes": 640, + "challenge_failure_count": 327, + "challenge_success_count": 313, + "challenge_success_fraction": 0.4890625, + "critic_contribution_value_prediction_corr": 0.9999999999987569, + "decoder_distance_residual_corr": 0.12395745146414912, + "mean_abs_raw_soma_corr": 0.9991696218086317, + "mean_abs_residual_soma_corr": 0.059213400745465807, + "mean_critic_expectedness_contribution": 0.3219422971700101, + "nonterminal_training_events": 12498, + "raw_minus_residual_abs_soma_corr": 0.9399562210631659, + "role_aligned_error_cv_corr": 0.36448618414523176, + "role_aligned_velocity_cv_corr": 0.9989099434588206, + "surrounding_event_decoder_balanced_acc": 0.5486811231901549, + "target_success_fraction": { + "1.7077688932418824": 0.7890625, + "1.7300113141536713": 0.5859375, + "1.74697607755661": 0.4140625, + "1.7630531013011932": 0.421875, + "1.7839327335357666": 0.234375 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.39742670634277316, + "terminal_previous_soma_outcome_balanced_acc": 0.7781995290715283, + "terminal_residual_minus_previous_soma_acc": 0.22180047092847166, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.39543003501629415, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6344237593135889 + }, + "split": "untouched_confirmation", + "wall_s": 2.514986202120781, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603000, + "role_cosine_after_warmup": 0.9941118955612183, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603000, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603000, + "role_cosine_after_warmup": 0.9941118955612183, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603000, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603000, + "role_cosine_after_warmup": 0.9941118955612183, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603000, + "role_cosine_after_warmup": 0.9941118955612183, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m1.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m1.json new file mode 100644 index 0000000..6e8fcd7 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m1.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 1, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 30 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13278, + "acute_outcome_lesion": 13278, + "intact": 13278 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.9548203945159912, + "median": 1.8720353841781616, + "minimum": 1.6812132596969604 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600030, + "targets": [ + 1.8308271884918212, + 1.852718323469162, + 1.8721278309822083, + 1.888966155052185, + 1.9044726371765137 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.8308271884918212, + 1.852718323469162, + 1.8721278309822083, + 1.888966155052185, + 1.9044726371765137 + ], + "trajectory_seeds": { + "1": 640000, + "2": 640001, + "3": 640002, + "4": 640003, + "5": 640004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590030 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 10849, + "cursor_scalar_observations": 6176, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3088, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.03125, + 0.25, + 0.65625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9886317253112793, + "training_wall_s": 0.15590760111808777 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06517095118761063, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.006917059421539307, + "training_wall_s": 0.21150564029812813 + }, + "intact": { + "cost": { + "active_state_episode_steps": 9535, + "cursor_scalar_observations": 5606, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2803, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5485566854476929, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.15625, + 0.625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 257, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9882871508598328, + "training_wall_s": 0.2410336211323738 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 9158, + "cursor_scalar_observations": 5416, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2708, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.536676824092865, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.1875, + 0.75, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.1589844711124897 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 15858, + "cursor_scalar_observations": 8308, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4154, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.18737183511257172, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.078125, + 0.203125, + 0.34375, + 0.609375, + 0.625, + 0.8125, + 0.921875, + 0.96875, + 0.984375, + 1.0, + 0.984375 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 1158, + "final_success": 0.9921875, + "late_success": 0.9895833333333334, + "learning_gain": 0.984375, + "role_cosine_after_training": 0.9413983225822449, + "training_wall_s": 0.20670033618807793 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06536653637886047, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.995029866695404, + "training_wall_s": 0.18543066084384918 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 942.64453125, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 1, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 30 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.48041979949874686, + "acute_outcome_lesion_role_aligned_separation": -7.046035766317127e-05, + "causal_role_sign_inversion_index": 0.04284599327224807, + "challenge_episodes": 640, + "challenge_failure_count": 336, + "challenge_success_count": 304, + "challenge_success_fraction": 0.475, + "critic_contribution_value_prediction_corr": 0.9999999999992322, + "decoder_distance_residual_corr": 0.11800025707551332, + "mean_abs_raw_soma_corr": 0.9989441606327615, + "mean_abs_residual_soma_corr": 0.06723418385975527, + "mean_critic_expectedness_contribution": 0.30660154433455283, + "nonterminal_training_events": 8639, + "raw_minus_residual_abs_soma_corr": 0.9317099767730062, + "role_aligned_error_cv_corr": 0.33395313554369577, + "role_aligned_velocity_cv_corr": 0.998363650023106, + "surrounding_event_decoder_balanced_acc": 0.5379391167947509, + "target_success_fraction": { + "1.8308271884918212": 0.8046875, + "1.852718323469162": 0.6328125, + "1.8721278309822083": 0.4296875, + "1.888966155052185": 0.359375, + "1.9044726371765137": 0.1484375 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.3947077639788761, + "terminal_previous_soma_outcome_balanced_acc": 0.7212562656641603, + "terminal_residual_minus_previous_soma_acc": 0.2757675438596492, + "terminal_residual_outcome_balanced_acc": 0.9970238095238095, + "terminal_role_aligned_outcome_separation": 0.3946373036212129, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6644105144794102 + }, + "split": "untouched_confirmation", + "wall_s": 2.18957032635808, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603001, + "role_cosine_after_warmup": 0.9941653609275818, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603001, + "role_cosine_after_warmup": 0.006917059421539307, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603001, + "role_cosine_after_warmup": 0.9941653609275818, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603001, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603001, + "role_cosine_after_warmup": 0.9941653609275818, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603001, + "role_cosine_after_warmup": 0.9941653609275818, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m2.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m2.json new file mode 100644 index 0000000..241aeb1 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m2.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 2, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 30 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13232, + "acute_outcome_lesion": 13232, + "intact": 13232 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8825325965881348, + "median": 1.7805328369140625, + "minimum": 1.6139979362487793 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600030, + "targets": [ + 1.7380641460418702, + 1.764169603586197, + 1.7806350588798523, + 1.7947718381881714, + 1.8090749263763428 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7380641460418702, + 1.764169603586197, + 1.7806350588798523, + 1.7947718381881714, + 1.8090749263763428 + ], + "trajectory_seeds": { + "1": 640000, + "2": 640001, + "3": 640002, + "4": 640003, + "5": 640004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590030 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 11098, + "cursor_scalar_observations": 6300, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3150, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.015625, + 0.171875, + 0.6875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9903352856636047, + "training_wall_s": 0.1756497211754322 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.07240072637796402, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.09573888778686523, + "training_wall_s": 0.22359507158398628 + }, + "intact": { + "cost": { + "active_state_episode_steps": 9079, + "cursor_scalar_observations": 5384, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2692, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5689617991447449, + "daily_success": [ + 0.0, + 0.0, + 0.046875, + 0.140625, + 0.796875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.015625, + "evaluation_active_state_episode_steps": 257, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.984375, + "role_cosine_after_training": 0.9901400208473206, + "training_wall_s": 0.2396928369998932 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 7802, + "cursor_scalar_observations": 4832, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2416, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5762484073638916, + "daily_success": [ + 0.0, + 0.015625, + 0.203125, + 0.640625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.07291666666666667, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9270833333333334, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.17397300899028778 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 15614, + "cursor_scalar_observations": 8208, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4104, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.30080774426460266, + "daily_success": [ + 0.0, + 0.0, + 0.046875, + 0.03125, + 0.1875, + 0.453125, + 0.65625, + 0.828125, + 0.875, + 0.921875, + 0.9375, + 0.953125, + 0.9375, + 1.0 + ], + "early_success": 0.015625, + "evaluation_active_state_episode_steps": 1043, + "final_success": 0.99609375, + "late_success": 0.9635416666666666, + "learning_gain": 0.9479166666666666, + "role_cosine_after_training": 0.9109129905700684, + "training_wall_s": 0.23295319080352783 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.07324913889169693, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951210021972656, + "training_wall_s": 0.20577557384967804 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 941.21875, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 2, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 30 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5653345833821634, + "acute_outcome_lesion_role_aligned_separation": -0.00801669590307541, + "causal_role_sign_inversion_index": 0.04398860253956989, + "challenge_episodes": 640, + "challenge_failure_count": 318, + "challenge_success_count": 322, + "challenge_success_fraction": 0.503125, + "critic_contribution_value_prediction_corr": 0.9999999999997197, + "decoder_distance_residual_corr": 0.09350586003483317, + "mean_abs_raw_soma_corr": 0.9984743514176625, + "mean_abs_residual_soma_corr": 0.07147933672711584, + "mean_critic_expectedness_contribution": 0.4068592197853116, + "nonterminal_training_events": 8183, + "raw_minus_residual_abs_soma_corr": 0.9269950146905467, + "role_aligned_error_cv_corr": 0.36689895391161137, + "role_aligned_velocity_cv_corr": 0.997529342641134, + "surrounding_event_decoder_balanced_acc": 0.5368580548559236, + "target_success_fraction": { + "1.7380641460418702": 0.7578125, + "1.764169603586197": 0.671875, + "1.7806350588798523": 0.4921875, + "1.7947718381881714": 0.3359375, + "1.8090749263763428": 0.2578125 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.4150091659778833, + "terminal_previous_soma_outcome_balanced_acc": 0.7463377475682644, + "terminal_residual_minus_previous_soma_acc": 0.25366225243173557, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.4069924700748079, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6306303887295226 + }, + "split": "untouched_confirmation", + "wall_s": 2.2932631336152554, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603002, + "role_cosine_after_warmup": 0.9943915605545044, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603002, + "role_cosine_after_warmup": -0.09573888778686523, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603002, + "role_cosine_after_warmup": 0.9943915605545044, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603002, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603002, + "role_cosine_after_warmup": 0.9943915605545044, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603002, + "role_cosine_after_warmup": 0.9943915605545044, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m3.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m3.json new file mode 100644 index 0000000..5dea392 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m3.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 3, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 30 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12612, + "acute_outcome_lesion": 12612, + "intact": 12612 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.9879597425460815, + "median": 1.9401476383209229, + "minimum": 1.718249797821045 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600030, + "targets": [ + 1.9072782039642333, + 1.9271499276161195, + 1.9401918649673462, + 1.9504702866077424, + 1.9608474731445313 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.9072782039642333, + 1.9271499276161195, + 1.9401918649673462, + 1.9504702866077424, + 1.9608474731445313 + ], + "trajectory_seeds": { + "1": 640000, + "2": 640001, + "3": 640002, + "4": 640003, + "5": 640004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590030 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 7996, + "cursor_scalar_observations": 4926, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2463, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.015625, + 0.0, + 0.046875, + 0.546875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.020833333333333332, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9791666666666666, + "role_cosine_after_training": 0.989193856716156, + "training_wall_s": 0.14595860615372658 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06709985435009003, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.006882439833134413, + "training_wall_s": 0.2109762355685234 + }, + "intact": { + "cost": { + "active_state_episode_steps": 8014, + "cursor_scalar_observations": 4872, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2436, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.7296893000602722, + "daily_success": [ + 0.015625, + 0.0, + 0.09375, + 0.703125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.036458333333333336, + "evaluation_active_state_episode_steps": 269, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9635416666666666, + "role_cosine_after_training": 0.9358880519866943, + "training_wall_s": 0.24309856444597244 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 7797, + "cursor_scalar_observations": 4798, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2399, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.7292206287384033, + "daily_success": [ + 0.015625, + 0.015625, + 0.125, + 0.75, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.052083333333333336, + "evaluation_active_state_episode_steps": 265, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9479166666666666, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.16316989436745644 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 14226, + "cursor_scalar_observations": 7580, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3790, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.30727362632751465, + "daily_success": [ + 0.015625, + 0.0, + 0.046875, + 0.15625, + 0.328125, + 0.609375, + 0.71875, + 0.890625, + 0.953125, + 0.921875, + 0.984375, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.020833333333333332, + "evaluation_active_state_episode_steps": 1104, + "final_success": 0.984375, + "late_success": 1.0, + "learning_gain": 0.9791666666666666, + "role_cosine_after_training": 0.9510527849197388, + "training_wall_s": 0.2021913081407547 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25085, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06687149405479431, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9950109124183655, + "training_wall_s": 0.18471171706914902 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 934.5546875, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 3, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 30 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.6661031566361881, + "acute_outcome_lesion_role_aligned_separation": 0.0605230507447061, + "causal_role_sign_inversion_index": 0.04508182951852431, + "challenge_episodes": 640, + "challenge_failure_count": 301, + "challenge_success_count": 339, + "challenge_success_fraction": 0.5296875, + "critic_contribution_value_prediction_corr": 0.9999999999999133, + "decoder_distance_residual_corr": 0.17845974840441675, + "mean_abs_raw_soma_corr": 0.9988264031253895, + "mean_abs_residual_soma_corr": 0.07095473246988797, + "mean_critic_expectedness_contribution": 0.18006295531237942, + "nonterminal_training_events": 7118, + "raw_minus_residual_abs_soma_corr": 0.9278716706555016, + "role_aligned_error_cv_corr": 0.39016816777870406, + "role_aligned_velocity_cv_corr": 0.9912536032650077, + "surrounding_event_decoder_balanced_acc": 0.5465429315869199, + "target_success_fraction": { + "1.9072782039642333": 0.78125, + "1.9271499276161195": 0.7578125, + "1.9401918649673462": 0.5, + "1.9504702866077424": 0.3828125, + "1.9608474731445313": 0.2265625 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.3726065484082109, + "terminal_previous_soma_outcome_balanced_acc": 0.7195876086594342, + "terminal_residual_minus_previous_soma_acc": 0.2770901322043532, + "terminal_residual_outcome_balanced_acc": 0.9966777408637874, + "terminal_role_aligned_outcome_separation": 0.433129599152917, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6010854354863036 + }, + "split": "untouched_confirmation", + "wall_s": 2.130977977067232, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603003, + "role_cosine_after_warmup": 0.9950518608093262, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603003, + "role_cosine_after_warmup": -0.006882439833134413, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603003, + "role_cosine_after_warmup": 0.9950518608093262, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603003, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603003, + "role_cosine_after_warmup": 0.9950518608093262, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603003, + "role_cosine_after_warmup": 0.9950518608093262, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m4.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m4.json new file mode 100644 index 0000000..6bd2957 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m4.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 4, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 30 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13215, + "acute_outcome_lesion": 13215, + "intact": 13215 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8303825855255127, + "median": 1.741257667541504, + "minimum": 1.5877881050109863 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600030, + "targets": [ + 1.7081695079803467, + 1.7258692026138305, + 1.7412704229354858, + 1.7550857603549956, + 1.7766520977020264 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7081695079803467, + 1.7258692026138305, + 1.7412704229354858, + 1.7550857603549956, + 1.7766520977020264 + ], + "trajectory_seeds": { + "1": 640000, + "2": 640001, + "3": 640002, + "4": 640003, + "5": 640004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590030 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.00390625, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.994904637336731, + "training_wall_s": 0.19335667416453362 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06304578483104706, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.01729092001914978, + "training_wall_s": 0.21148181706666946 + }, + "intact": { + "cost": { + "active_state_episode_steps": 15318, + "cursor_scalar_observations": 8146, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4073, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.4966675341129303, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.09375, + 0.34375, + 0.890625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9880502223968506, + "training_wall_s": 0.26330842077732086 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 14402, + "cursor_scalar_observations": 7720, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3860, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.6730367541313171, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.21875, + 0.71875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 276, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.1817156746983528 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 20250, + "cursor_scalar_observations": 10336, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 5168, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.18854676187038422, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.046875, + 0.140625, + 0.265625, + 0.5, + 0.59375, + 0.859375, + 0.921875, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 1036, + "final_success": 0.99609375, + "late_success": 0.9270833333333334, + "learning_gain": 0.9270833333333334, + "role_cosine_after_training": 0.9831609129905701, + "training_wall_s": 0.2081509679555893 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06297878175973892, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949046969413757, + "training_wall_s": 0.18322975561022758 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 957.484375, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 4, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 30 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5568627450980392, + "acute_outcome_lesion_role_aligned_separation": -0.0031391741738885925, + "causal_role_sign_inversion_index": 0.03850939637140704, + "challenge_episodes": 640, + "challenge_failure_count": 340, + "challenge_success_count": 300, + "challenge_success_fraction": 0.46875, + "critic_contribution_value_prediction_corr": 0.9999999999992066, + "decoder_distance_residual_corr": 0.1242231005538749, + "mean_abs_raw_soma_corr": 0.9992207488476954, + "mean_abs_residual_soma_corr": 0.05391572049406791, + "mean_critic_expectedness_contribution": 0.3511262845974606, + "nonterminal_training_events": 14422, + "raw_minus_residual_abs_soma_corr": 0.9453050283536275, + "role_aligned_error_cv_corr": 0.34408113359600123, + "role_aligned_velocity_cv_corr": 0.9990124504602647, + "surrounding_event_decoder_balanced_acc": 0.5432439177954838, + "target_success_fraction": { + "1.7081695079803467": 0.828125, + "1.7258692026138305": 0.6015625, + "1.7412704229354858": 0.4609375, + "1.7550857603549956": 0.328125, + "1.7766520977020264": 0.125 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.38312666738718737, + "terminal_previous_soma_outcome_balanced_acc": 0.7311764705882353, + "terminal_residual_minus_previous_soma_acc": 0.2688235294117647, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.3799874932132988, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6549313168642634 + }, + "split": "untouched_confirmation", + "wall_s": 2.4723090641200542, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603004, + "role_cosine_after_warmup": 0.9940178394317627, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603004, + "role_cosine_after_warmup": 0.01729092001914978, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603004, + "role_cosine_after_warmup": 0.9940178394317627, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603004, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603004, + "role_cosine_after_warmup": 0.9940178394317627, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603004, + "role_cosine_after_warmup": 0.9940178394317627, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m0.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m0.json new file mode 100644 index 0000000..116e040 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m0.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 0, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 31 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13344, + "acute_outcome_lesion": 13344, + "intact": 13344 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8499257564544678, + "median": 1.7456207275390625, + "minimum": 1.5906145572662354 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600031, + "targets": [ + 1.6990792751312256, + 1.723470389842987, + 1.7456920742988586, + 1.7613899171352387, + 1.7839894771575928 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.6990792751312256, + 1.723470389842987, + 1.7456920742988586, + 1.7613899171352387, + 1.7839894771575928 + ], + "trajectory_seeds": { + "1": 641000, + "2": 641001, + "3": 641002, + "4": 641003, + "5": 641004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590031 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 17773, + "cursor_scalar_observations": 9256, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4628, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.015625, + 0.015625, + 0.03125, + 0.046875, + 0.25, + 0.671875, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 264, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9349561333656311, + "training_wall_s": 0.19894864410161972 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06114146113395691, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.23242463544011116 + }, + "intact": { + "cost": { + "active_state_episode_steps": 13096, + "cursor_scalar_observations": 7164, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3582, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.48224231600761414, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.03125, + 0.0625, + 0.171875, + 0.484375, + 0.96875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 259, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9925675392150879, + "training_wall_s": 0.26290469616651535 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 13338, + "cursor_scalar_observations": 7268, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3634, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.6277045011520386, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.03125, + 0.0625, + 0.1875, + 0.5625, + 0.96875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 268, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.20504775270819664 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 19835, + "cursor_scalar_observations": 10148, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 5074, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.1494864970445633, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.015625, + 0.015625, + 0.078125, + 0.109375, + 0.40625, + 0.578125, + 0.75, + 0.828125, + 0.984375, + 0.875 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 1903, + "final_success": 0.96484375, + "late_success": 0.8958333333333334, + "learning_gain": 0.8958333333333334, + "role_cosine_after_training": 0.966598629951477, + "training_wall_s": 0.2313637211918831 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06195517256855965, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9957271218299866, + "training_wall_s": 0.18905282393097878 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 953.94921875, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 0, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 31 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5532193709904554, + "acute_outcome_lesion_role_aligned_separation": -0.0037801497200039558, + "causal_role_sign_inversion_index": 0.041016903323531806, + "challenge_episodes": 640, + "challenge_failure_count": 332, + "challenge_success_count": 308, + "challenge_success_fraction": 0.48125, + "critic_contribution_value_prediction_corr": 0.9999999999991654, + "decoder_distance_residual_corr": 0.11550503572047988, + "mean_abs_raw_soma_corr": 0.9990623209751222, + "mean_abs_residual_soma_corr": 0.06156255595114307, + "mean_critic_expectedness_contribution": 0.3246105738973889, + "nonterminal_training_events": 12200, + "raw_minus_residual_abs_soma_corr": 0.9374997650239791, + "role_aligned_error_cv_corr": 0.3619902737327908, + "role_aligned_velocity_cv_corr": 0.9990748248911435, + "surrounding_event_decoder_balanced_acc": 0.5502522748000614, + "target_success_fraction": { + "1.6990792751312256": 0.7578125, + "1.723470389842987": 0.625, + "1.7456920742988586": 0.4765625, + "1.7613899171352387": 0.3046875, + "1.7839894771575928": 0.2421875 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.38681074790839, + "terminal_previous_soma_outcome_balanced_acc": 0.7511930840244093, + "terminal_residual_minus_previous_soma_acc": 0.24880691597559068, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.38303059818838603, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6370845511583527 + }, + "split": "untouched_confirmation", + "wall_s": 2.5019032321870327, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603100, + "role_cosine_after_warmup": 0.9943863153457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603100, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603100, + "role_cosine_after_warmup": 0.9943863153457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603100, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603100, + "role_cosine_after_warmup": 0.9943863153457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603100, + "role_cosine_after_warmup": 0.9943863153457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m1.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m1.json new file mode 100644 index 0000000..fca0b6b --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m1.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 1, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 31 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12779, + "acute_outcome_lesion": 12779, + "intact": 12779 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8718771934509277, + "median": 1.7704464197158813, + "minimum": 1.631404995918274 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600031, + "targets": [ + 1.7303641319274903, + 1.7563523769378662, + 1.770524263381958, + 1.78693887591362, + 1.8043602466583253 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7303641319274903, + 1.7563523769378662, + 1.770524263381958, + 1.78693887591362, + 1.8043602466583253 + ], + "trajectory_seeds": { + "1": 641000, + "2": 641001, + "3": 641002, + "4": 641003, + "5": 641004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590031 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 10556, + "cursor_scalar_observations": 6042, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3021, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.03125, + 0.234375, + 0.90625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9887308478355408, + "training_wall_s": 0.15609203279018402 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06721100211143494, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.006917059421539307, + "training_wall_s": 0.21439224854111671 + }, + "intact": { + "cost": { + "active_state_episode_steps": 10039, + "cursor_scalar_observations": 5814, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2907, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5664430260658264, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.078125, + 0.46875, + 0.984375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9745795726776123, + "training_wall_s": 0.24971940368413925 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 9579, + "cursor_scalar_observations": 5616, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2808, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5435791611671448, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.171875, + 0.609375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.1637950874865055 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 15393, + "cursor_scalar_observations": 8100, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4050, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.3068060874938965, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.046875, + 0.15625, + 0.28125, + 0.53125, + 0.703125, + 0.953125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 1167, + "final_success": 0.98828125, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9411593675613403, + "training_wall_s": 0.20864040777087212 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06739386916160583, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9957154393196106, + "training_wall_s": 0.18799197673797607 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 943.484375, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 1, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 31 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5798249982909989, + "acute_outcome_lesion_role_aligned_separation": -0.008349476922578314, + "causal_role_sign_inversion_index": 0.04284164481366171, + "challenge_episodes": 640, + "challenge_failure_count": 321, + "challenge_success_count": 319, + "challenge_success_fraction": 0.4984375, + "critic_contribution_value_prediction_corr": 0.9999999999994682, + "decoder_distance_residual_corr": 0.1184867323551763, + "mean_abs_raw_soma_corr": 0.9988734944939421, + "mean_abs_residual_soma_corr": 0.06424228662255252, + "mean_critic_expectedness_contribution": 0.403613685737661, + "nonterminal_training_events": 9143, + "raw_minus_residual_abs_soma_corr": 0.9346312078713896, + "role_aligned_error_cv_corr": 0.3664888118346767, + "role_aligned_velocity_cv_corr": 0.9988354304545863, + "surrounding_event_decoder_balanced_acc": 0.5435065680168163, + "target_success_fraction": { + "1.7303641319274903": 0.8203125, + "1.7563523769378662": 0.65625, + "1.770524263381958": 0.4765625, + "1.78693887591362": 0.40625, + "1.8043602466583253": 0.1328125 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.4018920300804501, + "terminal_previous_soma_outcome_balanced_acc": 0.7331175109131924, + "terminal_residual_minus_previous_soma_acc": 0.26688248908680756, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.3935425531578718, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6323466186199096 + }, + "split": "untouched_confirmation", + "wall_s": 2.2389948591589928, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603101, + "role_cosine_after_warmup": 0.9948646426200867, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603101, + "role_cosine_after_warmup": 0.006917059421539307, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603101, + "role_cosine_after_warmup": 0.9948646426200867, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603101, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603101, + "role_cosine_after_warmup": 0.9948646426200867, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603101, + "role_cosine_after_warmup": 0.9948646426200867, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m2.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m2.json new file mode 100644 index 0000000..4e81653 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m2.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 2, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 31 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12444, + "acute_outcome_lesion": 12444, + "intact": 12444 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8338556289672852, + "median": 1.7279796600341797, + "minimum": 1.5054033994674683 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600031, + "targets": [ + 1.6913140535354614, + 1.7105235993862151, + 1.728020429611206, + 1.7467196226119994, + 1.7665327548980714 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.6913140535354614, + 1.7105235993862151, + 1.728020429611206, + 1.7467196226119994, + 1.7665327548980714 + ], + "trajectory_seeds": { + "1": 641000, + "2": 641001, + "3": 641002, + "4": 641003, + "5": 641004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590031 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 22917, + "cursor_scalar_observations": 11554, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 5777, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.015625, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0, + 0.015625, + 0.078125, + 0.046875, + 0.1875, + 0.734375, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 380, + "final_success": 1.0, + "late_success": 0.640625, + "learning_gain": 0.6354166666666666, + "role_cosine_after_training": 0.9704662561416626, + "training_wall_s": 0.2071349136531353 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25082, + "cursor_scalar_observations": 12542, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6271, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.07262592017650604, + "daily_success": [ + 0.0, + 0.015625, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": -0.005208333333333333, + "role_cosine_after_training": -0.09573888778686523, + "training_wall_s": 0.23077302798628807 + }, + "intact": { + "cost": { + "active_state_episode_steps": 11746, + "cursor_scalar_observations": 6562, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3281, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5240932106971741, + "daily_success": [ + 0.0, + 0.015625, + 0.0, + 0.0, + 0.0625, + 0.515625, + 0.921875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 257, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.992520809173584, + "training_wall_s": 0.2641731910407543 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 11285, + "cursor_scalar_observations": 6380, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3190, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.505804181098938, + "daily_success": [ + 0.0, + 0.015625, + 0.0, + 0.0, + 0.078125, + 0.671875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 257, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.18304626643657684 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 17139, + "cursor_scalar_observations": 8906, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4453, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.2952084243297577, + "daily_success": [ + 0.0, + 0.015625, + 0.0, + 0.0, + 0.015625, + 0.09375, + 0.34375, + 0.5625, + 0.671875, + 0.734375, + 0.96875, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 827, + "final_success": 0.99609375, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9332522749900818, + "training_wall_s": 0.2239033430814743 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25082, + "cursor_scalar_observations": 12542, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6271, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.07343900948762894, + "daily_success": [ + 0.0, + 0.015625, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": -0.005208333333333333, + "role_cosine_after_training": 0.9956480264663696, + "training_wall_s": 0.20438788831233978 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 948.23046875, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 2, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 31 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.6053790841926435, + "acute_outcome_lesion_role_aligned_separation": 0.003371551194363609, + "causal_role_sign_inversion_index": 0.045323503148447575, + "challenge_episodes": 640, + "challenge_failure_count": 286, + "challenge_success_count": 354, + "challenge_success_fraction": 0.553125, + "critic_contribution_value_prediction_corr": 0.9999999999995355, + "decoder_distance_residual_corr": 0.09154031579785411, + "mean_abs_raw_soma_corr": 0.9984311919019998, + "mean_abs_residual_soma_corr": 0.07011313249162582, + "mean_critic_expectedness_contribution": 0.3546685105035511, + "nonterminal_training_events": 10850, + "raw_minus_residual_abs_soma_corr": 0.9283180594103739, + "role_aligned_error_cv_corr": 0.36477872287299773, + "role_aligned_velocity_cv_corr": 0.9987356976354365, + "surrounding_event_decoder_balanced_acc": 0.5380200513033967, + "target_success_fraction": { + "1.6913140535354614": 0.84375, + "1.7105235993862151": 0.6484375, + "1.728020429611206": 0.546875, + "1.7467196226119994": 0.4296875, + "1.7665327548980714": 0.296875 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.40253520722011477, + "terminal_previous_soma_outcome_balanced_acc": 0.7287246651653432, + "terminal_residual_minus_previous_soma_acc": 0.2712753348346568, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.4059067584144784, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6339569747624387 + }, + "split": "untouched_confirmation", + "wall_s": 2.422639410942793, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603102, + "role_cosine_after_warmup": 0.9946048259735107, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603102, + "role_cosine_after_warmup": -0.09573888778686523, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603102, + "role_cosine_after_warmup": 0.9946048259735107, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603102, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603102, + "role_cosine_after_warmup": 0.9946048259735107, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603102, + "role_cosine_after_warmup": 0.9946048259735107, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m3.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m3.json new file mode 100644 index 0000000..2156303 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m3.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 3, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 31 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12189, + "acute_outcome_lesion": 12189, + "intact": 12189 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.946171522140503, + "median": 1.8184009790420532, + "minimum": 1.528581142425537 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600031, + "targets": [ + 1.768526291847229, + 1.7987976253032685, + 1.818694531917572, + 1.8393855273723603, + 1.8587792634963989 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.768526291847229, + 1.7987976253032685, + 1.818694531917572, + 1.8393855273723603, + 1.8587792634963989 + ], + "trajectory_seeds": { + "1": 641000, + "2": 641001, + "3": 641002, + "4": 641003, + "5": 641004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590031 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 10209, + "cursor_scalar_observations": 5910, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2955, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.03125, + 0.296875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9886515140533447, + "training_wall_s": 0.15481916069984436 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06708750873804092, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.006882439833134413, + "training_wall_s": 0.21405809745192528 + }, + "intact": { + "cost": { + "active_state_episode_steps": 9529, + "cursor_scalar_observations": 5576, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2788, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5406008958816528, + "daily_success": [ + 0.0, + 0.0, + 0.03125, + 0.125, + 0.65625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 263, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 0.9934541583061218, + "training_wall_s": 0.24109508842229843 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 9448, + "cursor_scalar_observations": 5538, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2769, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5557917952537537, + "daily_success": [ + 0.0, + 0.0, + 0.03125, + 0.125, + 0.671875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.16011293977499008 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 15005, + "cursor_scalar_observations": 7940, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3970, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.32433053851127625, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.03125, + 0.203125, + 0.328125, + 0.578125, + 0.734375, + 0.9375, + 0.890625, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 1063, + "final_success": 0.98828125, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.8994001746177673, + "training_wall_s": 0.20385408401489258 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06761886179447174, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.99561607837677, + "training_wall_s": 0.1856268048286438 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 940.69921875, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 3, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 31 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5830049261083744, + "acute_outcome_lesion_role_aligned_separation": 0.0008006474732861757, + "causal_role_sign_inversion_index": 0.04285774438964318, + "challenge_episodes": 640, + "challenge_failure_count": 290, + "challenge_success_count": 350, + "challenge_success_fraction": 0.546875, + "critic_contribution_value_prediction_corr": 0.9999999999990883, + "decoder_distance_residual_corr": 0.1502318590258936, + "mean_abs_raw_soma_corr": 0.9990027384396377, + "mean_abs_residual_soma_corr": 0.06792842488706131, + "mean_critic_expectedness_contribution": 0.3605008679854347, + "nonterminal_training_events": 8633, + "raw_minus_residual_abs_soma_corr": 0.9310743135525764, + "role_aligned_error_cv_corr": 0.36684157797403466, + "role_aligned_velocity_cv_corr": 0.9984266062582129, + "surrounding_event_decoder_balanced_acc": 0.5483350640845428, + "target_success_fraction": { + "1.768526291847229": 0.78125, + "1.7987976253032685": 0.734375, + "1.818694531917572": 0.5625, + "1.8393855273723603": 0.3671875, + "1.8587792634963989": 0.2890625 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.39354691332329245, + "terminal_previous_soma_outcome_balanced_acc": 0.734088669950739, + "terminal_residual_minus_previous_soma_acc": 0.2641871921182266, + "terminal_residual_outcome_balanced_acc": 0.9982758620689656, + "terminal_role_aligned_outcome_separation": 0.3943475607965786, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6315850282841782 + }, + "split": "untouched_confirmation", + "wall_s": 2.1836028955876827, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603103, + "role_cosine_after_warmup": 0.9947764277458191, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603103, + "role_cosine_after_warmup": -0.006882439833134413, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603103, + "role_cosine_after_warmup": 0.9947764277458191, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603103, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603103, + "role_cosine_after_warmup": 0.9947764277458191, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603103, + "role_cosine_after_warmup": 0.9947764277458191, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m4.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m4.json new file mode 100644 index 0000000..a9c07d1 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m4.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 4, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 31 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12481, + "acute_outcome_lesion": 12481, + "intact": 12481 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.9478094577789307, + "median": 1.848829746246338, + "minimum": 1.619896650314331 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600031, + "targets": [ + 1.8018213748931884, + 1.8293907046318054, + 1.8489220142364502, + 1.86823810338974, + 1.88649525642395 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.8018213748931884, + 1.8293907046318054, + 1.8489220142364502, + 1.86823810338974, + 1.88649525642395 + ], + "trajectory_seeds": { + "1": 641000, + "2": 641001, + "3": 641002, + "4": 641003, + "5": 641004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590031 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 24844, + "cursor_scalar_observations": 12440, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6220, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.015625, + 0.046875, + 0.015625, + 0.015625, + 0.03125, + 0.0625, + 0.203125 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 6100, + "final_success": 0.30078125, + "late_success": 0.09895833333333333, + "learning_gain": 0.09895833333333333, + "role_cosine_after_training": 0.9961150884628296, + "training_wall_s": 0.19386376440525055 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06359569728374481, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.01729092001914978, + "training_wall_s": 0.21059979870915413 + }, + "intact": { + "cost": { + "active_state_episode_steps": 14111, + "cursor_scalar_observations": 7612, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3806, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5349434614181519, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.078125, + 0.296875, + 0.796875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 266, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9691336750984192, + "training_wall_s": 0.26048726588487625 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 13843, + "cursor_scalar_observations": 7492, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3746, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.46775054931640625, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.109375, + 0.359375, + 0.84375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 259, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.17686181142926216 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 19214, + "cursor_scalar_observations": 9874, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4937, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.19615936279296875, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.046875, + 0.125, + 0.3125, + 0.53125, + 0.71875, + 0.8125, + 0.90625, + 0.953125, + 0.984375 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 1278, + "final_success": 0.9921875, + "late_success": 0.9479166666666666, + "learning_gain": 0.9479166666666666, + "role_cosine_after_training": 0.9871839880943298, + "training_wall_s": 0.20905012637376785 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06368452310562134, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.995797336101532, + "training_wall_s": 0.18488534912467003 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 958.44140625, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 4, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 31 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5529448621553885, + "acute_outcome_lesion_role_aligned_separation": 0.010989038408498214, + "causal_role_sign_inversion_index": 0.042290149677602265, + "challenge_episodes": 640, + "challenge_failure_count": 304, + "challenge_success_count": 336, + "challenge_success_fraction": 0.525, + "critic_contribution_value_prediction_corr": 0.9999999999992204, + "decoder_distance_residual_corr": 0.11739384988355923, + "mean_abs_raw_soma_corr": 0.9991035435064983, + "mean_abs_residual_soma_corr": 0.05746311806667046, + "mean_critic_expectedness_contribution": 0.17735057790482808, + "nonterminal_training_events": 13215, + "raw_minus_residual_abs_soma_corr": 0.9416404254398278, + "role_aligned_error_cv_corr": 0.3526399100393868, + "role_aligned_velocity_cv_corr": 0.998187060895449, + "surrounding_event_decoder_balanced_acc": 0.5431536751969013, + "target_success_fraction": { + "1.8018213748931884": 0.7734375, + "1.8293907046318054": 0.75, + "1.8489220142364502": 0.5703125, + "1.86823810338974": 0.3125, + "1.88649525642395": 0.21875 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.36886632282548903, + "terminal_previous_soma_outcome_balanced_acc": 0.6877349624060151, + "terminal_residual_minus_previous_soma_acc": 0.30897556390977443, + "terminal_residual_outcome_balanced_acc": 0.9967105263157895, + "terminal_role_aligned_outcome_separation": 0.3798553612339872, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6455471508560622 + }, + "split": "untouched_confirmation", + "wall_s": 2.4267357736825943, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603104, + "role_cosine_after_warmup": 0.9956888556480408, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603104, + "role_cosine_after_warmup": 0.01729092001914978, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603104, + "role_cosine_after_warmup": 0.9956888556480408, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603104, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603104, + "role_cosine_after_warmup": 0.9956888556480408, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603104, + "role_cosine_after_warmup": 0.9956888556480408, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m0.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m0.json new file mode 100644 index 0000000..46fea4a --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m0.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 0, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 32 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13095, + "acute_outcome_lesion": 13095, + "intact": 13095 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8425406217575073, + "median": 1.7279386520385742, + "minimum": 1.5847492218017578 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600032, + "targets": [ + 1.6887985467910767, + 1.7095706522464753, + 1.7279988527297974, + 1.7469618499279023, + 1.763807725906372 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.6887985467910767, + 1.7095706522464753, + 1.7279988527297974, + 1.7469618499279023, + 1.763807725906372 + ], + "trajectory_seeds": { + "1": 642000, + "2": 642001, + "3": 642002, + "4": 642003, + "5": 642004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590032 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7149, + "final_success": 0.00390625, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.996315062046051, + "training_wall_s": 0.19666767120361328 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.061198148876428604, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21383566036820412 + }, + "intact": { + "cost": { + "active_state_episode_steps": 14253, + "cursor_scalar_observations": 7686, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3843, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.46887069940567017, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.21875, + 0.78125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 258, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9764077067375183, + "training_wall_s": 0.2746609225869179 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 14100, + "cursor_scalar_observations": 7602, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3801, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.4751800000667572, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.046875, + 0.28125, + 0.796875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 258, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.177669707685709 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 19294, + "cursor_scalar_observations": 9874, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4937, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.1886383444070816, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.0625, + 0.203125, + 0.34375, + 0.59375, + 0.75, + 0.953125, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 1007, + "final_success": 0.9921875, + "late_success": 0.984375, + "learning_gain": 0.984375, + "role_cosine_after_training": 0.9549081921577454, + "training_wall_s": 0.2088504172861576 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.062172841280698776, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.996315062046051, + "training_wall_s": 0.1860225833952427 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 958.7734375, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 0, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 32 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5298156177975701, + "acute_outcome_lesion_role_aligned_separation": 0.0012222849014450476, + "causal_role_sign_inversion_index": 0.03892022098079853, + "challenge_episodes": 640, + "challenge_failure_count": 318, + "challenge_success_count": 322, + "challenge_success_fraction": 0.503125, + "critic_contribution_value_prediction_corr": 0.9999999999991087, + "decoder_distance_residual_corr": 0.1107168553685575, + "mean_abs_raw_soma_corr": 0.9991670340589252, + "mean_abs_residual_soma_corr": 0.05431112306501673, + "mean_critic_expectedness_contribution": 0.33374964823712994, + "nonterminal_training_events": 13357, + "raw_minus_residual_abs_soma_corr": 0.9448559109939084, + "role_aligned_error_cv_corr": 0.3518302283556285, + "role_aligned_velocity_cv_corr": 0.9987888448435762, + "surrounding_event_decoder_balanced_acc": 0.5439545109571295, + "target_success_fraction": { + "1.6887985467910767": 0.7578125, + "1.7095706522464753": 0.6796875, + "1.7279988527297974": 0.453125, + "1.7469618499279023": 0.40625, + "1.763807725906372": 0.21875 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.3941982488700615, + "terminal_previous_soma_outcome_balanced_acc": 0.709090198835892, + "terminal_residual_minus_previous_soma_acc": 0.29090980116410803, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.39542053377150654, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6469586164879477 + }, + "split": "untouched_confirmation", + "wall_s": 2.464668322354555, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603200, + "role_cosine_after_warmup": 0.9932844638824463, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603200, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603200, + "role_cosine_after_warmup": 0.9932844638824463, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603200, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603200, + "role_cosine_after_warmup": 0.9932844638824463, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603200, + "role_cosine_after_warmup": 0.9932844638824463, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m1.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m1.json new file mode 100644 index 0000000..ae5b352 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m1.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 1, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 32 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13197, + "acute_outcome_lesion": 13197, + "intact": 13197 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.839465856552124, + "median": 1.7447079420089722, + "minimum": 1.5286614894866943 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600032, + "targets": [ + 1.7064756870269775, + 1.7277086198329925, + 1.7448158860206604, + 1.7594623565673828, + 1.7777812480926514 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7064756870269775, + 1.7277086198329925, + 1.7448158860206604, + 1.7594623565673828, + 1.7777812480926514 + ], + "trajectory_seeds": { + "1": 642000, + "2": 642001, + "3": 642002, + "4": 642003, + "5": 642004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590032 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 8309, + "cursor_scalar_observations": 5052, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2526, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.015625, + 0.0625, + 0.359375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.026041666666666668, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9739583333333334, + "role_cosine_after_training": 0.9838115572929382, + "training_wall_s": 0.14876241609454155 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06551171839237213, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.006917059421539307, + "training_wall_s": 0.20988380163908005 + }, + "intact": { + "cost": { + "active_state_episode_steps": 8176, + "cursor_scalar_observations": 4970, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2485, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5582403540611267, + "daily_success": [ + 0.0, + 0.015625, + 0.078125, + 0.515625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.03125, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.96875, + "role_cosine_after_training": 0.9938780665397644, + "training_wall_s": 0.23583389073610306 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 8098, + "cursor_scalar_observations": 4942, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2471, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5624566078186035, + "daily_success": [ + 0.0, + 0.015625, + 0.078125, + 0.53125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.03125, + "evaluation_active_state_episode_steps": 257, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.96875, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.16362693533301353 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 14370, + "cursor_scalar_observations": 7642, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3821, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.3587868809700012, + "daily_success": [ + 0.0, + 0.015625, + 0.046875, + 0.125, + 0.34375, + 0.515625, + 0.625, + 0.84375, + 0.96875, + 0.96875, + 0.984375, + 0.984375, + 1.0, + 1.0 + ], + "early_success": 0.020833333333333332, + "evaluation_active_state_episode_steps": 552, + "final_success": 1.0, + "late_success": 0.9947916666666666, + "learning_gain": 0.9739583333333334, + "role_cosine_after_training": 0.8972746729850769, + "training_wall_s": 0.20119396969676018 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06581009179353714, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9962126016616821, + "training_wall_s": 0.1819687932729721 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 936.0859375, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 1, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 32 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5727208784338244, + "acute_outcome_lesion_role_aligned_separation": -0.0009131268129543013, + "causal_role_sign_inversion_index": 0.0448768071145361, + "challenge_episodes": 640, + "challenge_failure_count": 326, + "challenge_success_count": 314, + "challenge_success_fraction": 0.490625, + "critic_contribution_value_prediction_corr": 0.9999999999987023, + "decoder_distance_residual_corr": 0.11275966315486538, + "mean_abs_raw_soma_corr": 0.9987214042329219, + "mean_abs_residual_soma_corr": 0.07483543237113474, + "mean_critic_expectedness_contribution": 0.36799634712323687, + "nonterminal_training_events": 7280, + "raw_minus_residual_abs_soma_corr": 0.9238859718617871, + "role_aligned_error_cv_corr": 0.3891806866805502, + "role_aligned_velocity_cv_corr": 0.9981286331323391, + "surrounding_event_decoder_balanced_acc": 0.5415856723757944, + "target_success_fraction": { + "1.7064756870269775": 0.7578125, + "1.7277086198329925": 0.6484375, + "1.7448158860206604": 0.453125, + "1.7594623565673828": 0.390625, + "1.7777812480926514": 0.203125 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.39622390380398786, + "terminal_previous_soma_outcome_balanced_acc": 0.7454769254816147, + "terminal_residual_minus_previous_soma_acc": 0.2545230745183853, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.39531077699103356, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6089479464517888 + }, + "split": "untouched_confirmation", + "wall_s": 2.1287473514676094, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603201, + "role_cosine_after_warmup": 0.9919762015342712, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603201, + "role_cosine_after_warmup": 0.006917059421539307, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603201, + "role_cosine_after_warmup": 0.9919762015342712, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603201, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603201, + "role_cosine_after_warmup": 0.9919762015342712, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603201, + "role_cosine_after_warmup": 0.9919762015342712, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m2.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m2.json new file mode 100644 index 0000000..60195a0 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m2.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 2, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 32 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12931, + "acute_outcome_lesion": 12931, + "intact": 12931 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8897517919540405, + "median": 1.7930231094360352, + "minimum": 1.557706594467163 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600032, + "targets": [ + 1.7486443758010863, + 1.775669765472412, + 1.7931550741195679, + 1.8108321189880372, + 1.8302961826324462 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7486443758010863, + 1.775669765472412, + 1.7931550741195679, + 1.8108321189880372, + 1.8302961826324462 + ], + "trajectory_seeds": { + "1": 642000, + "2": 642001, + "3": 642002, + "4": 642003, + "5": 642004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590032 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 10163, + "cursor_scalar_observations": 5892, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2946, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.03125, + 0.3125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9920443892478943, + "training_wall_s": 0.17430328205227852 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25085, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.07042737305164337, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": -0.005208333333333333, + "role_cosine_after_training": -0.09573888778686523, + "training_wall_s": 0.21658989414572716 + }, + "intact": { + "cost": { + "active_state_episode_steps": 9802, + "cursor_scalar_observations": 5718, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2859, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5440401434898376, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.046875, + 0.484375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 258, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9878109097480774, + "training_wall_s": 0.24366315454244614 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 9799, + "cursor_scalar_observations": 5720, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2860, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5370547771453857, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.046875, + 0.484375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.18266911059617996 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 13945, + "cursor_scalar_observations": 7404, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3702, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.27621981501579285, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.03125, + 0.21875, + 0.609375, + 0.671875, + 0.875, + 0.9375, + 1.0, + 1.0, + 1.0, + 0.984375, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 1331, + "final_success": 0.98046875, + "late_success": 0.9947916666666666, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 0.9046207070350647, + "training_wall_s": 0.22472821548581123 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25085, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0711759701371193, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": -0.005208333333333333, + "role_cosine_after_training": 0.9962702989578247, + "training_wall_s": 0.19579441845417023 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 942.1328125, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 2, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 32 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5840983975205076, + "acute_outcome_lesion_role_aligned_separation": -0.008790232975897705, + "causal_role_sign_inversion_index": 0.044703932770676566, + "challenge_episodes": 640, + "challenge_failure_count": 309, + "challenge_success_count": 331, + "challenge_success_fraction": 0.5171875, + "critic_contribution_value_prediction_corr": 0.9999999999995718, + "decoder_distance_residual_corr": 0.11049020223221237, + "mean_abs_raw_soma_corr": 0.9984911157726121, + "mean_abs_residual_soma_corr": 0.07227685967132579, + "mean_critic_expectedness_contribution": 0.34219986837841365, + "nonterminal_training_events": 8906, + "raw_minus_residual_abs_soma_corr": 0.9262142561012863, + "role_aligned_error_cv_corr": 0.3542824671793922, + "role_aligned_velocity_cv_corr": 0.9991640013880079, + "surrounding_event_decoder_balanced_acc": 0.540257224751793, + "target_success_fraction": { + "1.7486443758010863": 0.796875, + "1.775669765472412": 0.671875, + "1.7931550741195679": 0.546875, + "1.8108321189880372": 0.3515625, + "1.8302961826324462": 0.21875 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.3765925358632958, + "terminal_previous_soma_outcome_balanced_acc": 0.7363339492955543, + "terminal_residual_minus_previous_soma_acc": 0.2636660507044457, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.36780230288739807, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6448815342086156 + }, + "split": "untouched_confirmation", + "wall_s": 2.3039599284529686, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603202, + "role_cosine_after_warmup": 0.9945364594459534, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603202, + "role_cosine_after_warmup": -0.09573888778686523, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603202, + "role_cosine_after_warmup": 0.9945364594459534, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603202, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603202, + "role_cosine_after_warmup": 0.9945364594459534, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603202, + "role_cosine_after_warmup": 0.9945364594459534, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m3.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m3.json new file mode 100644 index 0000000..d49f8a1 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m3.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 3, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 32 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12936, + "acute_outcome_lesion": 12936, + "intact": 12936 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8906066417694092, + "median": 1.7748355865478516, + "minimum": 1.6290333271026611 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600032, + "targets": [ + 1.7369842052459716, + 1.7613054513931274, + 1.7748485207557678, + 1.7908255577087402, + 1.8147287607192992 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7369842052459716, + 1.7613054513931274, + 1.7748485207557678, + 1.7908255577087402, + 1.8147287607192992 + ], + "trajectory_seeds": { + "1": 642000, + "2": 642001, + "3": 642002, + "4": 642003, + "5": 642004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590032 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 8326, + "cursor_scalar_observations": 5088, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2544, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.03125, + 0.34375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 0.9884672164916992, + "training_wall_s": 0.14801248162984848 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25079, + "cursor_scalar_observations": 12540, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6270, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06553865224123001, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.005208333333333333, + "learning_gain": 0.005208333333333333, + "role_cosine_after_training": -0.006882439833134413, + "training_wall_s": 0.22043022140860558 + }, + "intact": { + "cost": { + "active_state_episode_steps": 8289, + "cursor_scalar_observations": 5048, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2524, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5611994862556458, + "daily_success": [ + 0.0, + 0.0, + 0.03125, + 0.4375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 0.9723539352416992, + "training_wall_s": 0.23410236462950706 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 8230, + "cursor_scalar_observations": 5016, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2508, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5559794306755066, + "daily_success": [ + 0.0, + 0.0, + 0.03125, + 0.484375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.16238350421190262 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 15474, + "cursor_scalar_observations": 8170, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4085, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.299395352602005, + "daily_success": [ + 0.0, + 0.0, + 0.03125, + 0.15625, + 0.171875, + 0.328125, + 0.65625, + 0.859375, + 0.921875, + 0.890625, + 0.9375, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 784, + "final_success": 0.99609375, + "late_success": 1.0, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 0.9431062936782837, + "training_wall_s": 0.21385130658745766 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25079, + "cursor_scalar_observations": 12540, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6270, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06586109101772308, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7157, + "final_success": 0.0078125, + "late_success": 0.005208333333333333, + "learning_gain": 0.005208333333333333, + "role_cosine_after_training": 0.9962142109870911, + "training_wall_s": 0.1938561424612999 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 936.9296875, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 3, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 32 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5844393011650504, + "acute_outcome_lesion_role_aligned_separation": -0.0054868800776807225, + "causal_role_sign_inversion_index": 0.04282685276465098, + "challenge_episodes": 640, + "challenge_failure_count": 319, + "challenge_success_count": 321, + "challenge_success_fraction": 0.5015625, + "critic_contribution_value_prediction_corr": 0.9999999999994792, + "decoder_distance_residual_corr": 0.14779486236004197, + "mean_abs_raw_soma_corr": 0.9988550372755418, + "mean_abs_residual_soma_corr": 0.06102132627162356, + "mean_critic_expectedness_contribution": 0.4046042795350481, + "nonterminal_training_events": 7393, + "raw_minus_residual_abs_soma_corr": 0.9378337110039182, + "role_aligned_error_cv_corr": 0.38251224209011886, + "role_aligned_velocity_cv_corr": 0.9968255795268922, + "surrounding_event_decoder_balanced_acc": 0.5441025609227946, + "target_success_fraction": { + "1.7369842052459716": 0.765625, + "1.7613054513931274": 0.6328125, + "1.7748485207557678": 0.5390625, + "1.7908255577087402": 0.390625, + "1.8147287607192992": 0.1796875 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.40831209855047074, + "terminal_previous_soma_outcome_balanced_acc": 0.7560718366390297, + "terminal_residual_minus_previous_soma_acc": 0.24236076524184813, + "terminal_residual_outcome_balanced_acc": 0.9984326018808778, + "terminal_role_aligned_outcome_separation": 0.40282521847279, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6143133374367733 + }, + "split": "untouched_confirmation", + "wall_s": 2.176859501749277, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603203, + "role_cosine_after_warmup": 0.9955039620399475, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603203, + "role_cosine_after_warmup": -0.006882439833134413, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603203, + "role_cosine_after_warmup": 0.9955039620399475, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603203, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603203, + "role_cosine_after_warmup": 0.9955039620399475, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603203, + "role_cosine_after_warmup": 0.9955039620399475, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m4.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m4.json new file mode 100644 index 0000000..897fa75 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m4.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 4, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 32 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12590, + "acute_outcome_lesion": 12590, + "intact": 12590 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8256962299346924, + "median": 1.708701252937317, + "minimum": 1.585423469543457 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600032, + "targets": [ + 1.6746601104736327, + 1.695594596862793, + 1.7089820504188538, + 1.7232511579990386, + 1.7401185512542725 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.6746601104736327, + 1.695594596862793, + 1.7089820504188538, + 1.7232511579990386, + 1.7401185512542725 + ], + "trajectory_seeds": { + "1": 642000, + "2": 642001, + "3": 642002, + "4": 642003, + "5": 642004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590032 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 18517, + "cursor_scalar_observations": 9596, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4798, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.03125, + 0.046875, + 0.09375, + 0.4375, + 0.96875, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 260, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9896625280380249, + "training_wall_s": 0.177993044257164 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06390243023633957, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.01729092001914978, + "training_wall_s": 0.21118981018662453 + }, + "intact": { + "cost": { + "active_state_episode_steps": 12855, + "cursor_scalar_observations": 7062, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3531, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.47574204206466675, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.171875, + 0.625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9900267124176025, + "training_wall_s": 0.25169622153043747 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 11636, + "cursor_scalar_observations": 6522, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3261, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5253611207008362, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.078125, + 0.453125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 257, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.18100709468126297 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 18404, + "cursor_scalar_observations": 9508, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4754, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.22336269915103912, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.09375, + 0.1875, + 0.40625, + 0.5, + 0.75, + 0.890625, + 0.953125, + 0.984375, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 847, + "final_success": 0.99609375, + "late_success": 0.9791666666666666, + "learning_gain": 0.9791666666666666, + "role_cosine_after_training": 0.9733901619911194, + "training_wall_s": 0.2070014774799347 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06398437917232513, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9962295293807983, + "training_wall_s": 0.18490266054868698 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 954.83203125, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 4, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 32 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5195035460992907, + "acute_outcome_lesion_role_aligned_separation": 0.0011440974938929371, + "causal_role_sign_inversion_index": 0.04020785015000099, + "challenge_episodes": 640, + "challenge_failure_count": 282, + "challenge_success_count": 358, + "challenge_success_fraction": 0.559375, + "critic_contribution_value_prediction_corr": 0.9999999999985824, + "decoder_distance_residual_corr": 0.11402591987327763, + "mean_abs_raw_soma_corr": 0.9991357390073798, + "mean_abs_residual_soma_corr": 0.056203858190257264, + "mean_critic_expectedness_contribution": 0.3171189506993658, + "nonterminal_training_events": 11959, + "raw_minus_residual_abs_soma_corr": 0.9429318808171225, + "role_aligned_error_cv_corr": 0.35056782758067284, + "role_aligned_velocity_cv_corr": 0.9986312234017266, + "surrounding_event_decoder_balanced_acc": 0.5454484373262114, + "target_success_fraction": { + "1.6746601104736327": 0.828125, + "1.695594596862793": 0.6796875, + "1.7089820504188538": 0.59375, + "1.7232511579990386": 0.3984375, + "1.7401185512542725": 0.296875 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.39602820639246067, + "terminal_previous_soma_outcome_balanced_acc": 0.7157969808629503, + "terminal_residual_minus_previous_soma_acc": 0.2842030191370497, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.3971723038863536, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6480633958210538 + }, + "split": "untouched_confirmation", + "wall_s": 2.3565363064408302, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603204, + "role_cosine_after_warmup": 0.9960409998893738, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603204, + "role_cosine_after_warmup": 0.01729092001914978, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603204, + "role_cosine_after_warmup": 0.9960409998893738, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603204, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603204, + "role_cosine_after_warmup": 0.9960409998893738, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603204, + "role_cosine_after_warmup": 0.9960409998893738, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m0.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m0.json new file mode 100644 index 0000000..4dfd010 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m0.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 0, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 33 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13234, + "acute_outcome_lesion": 13234, + "intact": 13234 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.883805751800537, + "median": 1.7651097774505615, + "minimum": 1.5316593647003174 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600033, + "targets": [ + 1.7193473339080811, + 1.745363223552704, + 1.765318751335144, + 1.7825899004936219, + 1.805083966255188 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7193473339080811, + 1.745363223552704, + 1.765318751335144, + 1.7825899004936219, + 1.805083966255188 + ], + "trajectory_seeds": { + "1": 643000, + "2": 643001, + "3": 643002, + "4": 643003, + "5": 643004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590033 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 18473, + "cursor_scalar_observations": 9588, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4794, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.015625, + 0.0, + 0.0, + 0.125, + 0.546875, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 258, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9865919351577759, + "training_wall_s": 0.1779797337949276 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0610068142414093, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.20908086374402046 + }, + "intact": { + "cost": { + "active_state_episode_steps": 13019, + "cursor_scalar_observations": 7134, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3567, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5035372972488403, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.21875, + 0.5, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 261, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9897723197937012, + "training_wall_s": 0.25168194994330406 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 12906, + "cursor_scalar_observations": 7086, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3543, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.4940507113933563, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.21875, + 0.609375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 259, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.1717309206724167 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 19812, + "cursor_scalar_observations": 10144, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 5072, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.19735406339168549, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.0625, + 0.046875, + 0.296875, + 0.375, + 0.65625, + 0.734375, + 0.875, + 0.90625, + 0.984375 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 1547, + "final_success": 0.984375, + "late_success": 0.921875, + "learning_gain": 0.921875, + "role_cosine_after_training": 0.9774383902549744, + "training_wall_s": 0.20789314061403275 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06178329139947891, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9944860935211182, + "training_wall_s": 0.18246862664818764 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 949.91796875, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 0, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 33 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.6193277885827657, + "acute_outcome_lesion_role_aligned_separation": -0.014488894261373786, + "causal_role_sign_inversion_index": 0.039108251369288596, + "challenge_episodes": 640, + "challenge_failure_count": 329, + "challenge_success_count": 311, + "challenge_success_fraction": 0.4859375, + "critic_contribution_value_prediction_corr": 0.9999999999996395, + "decoder_distance_residual_corr": 0.11429122298040599, + "mean_abs_raw_soma_corr": 0.9991313061283691, + "mean_abs_residual_soma_corr": 0.055993864071252356, + "mean_critic_expectedness_contribution": 0.3264753368204509, + "nonterminal_training_events": 12123, + "raw_minus_residual_abs_soma_corr": 0.9431374420571167, + "role_aligned_error_cv_corr": 0.3563285721004109, + "role_aligned_velocity_cv_corr": 0.9992809696972655, + "surrounding_event_decoder_balanced_acc": 0.542198682141877, + "target_success_fraction": { + "1.7193473339080811": 0.78125, + "1.745363223552704": 0.6328125, + "1.765318751335144": 0.53125, + "1.7825899004936219": 0.296875, + "1.805083966255188": 0.1875 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.37893073372171326, + "terminal_previous_soma_outcome_balanced_acc": 0.7497532227641005, + "terminal_residual_minus_previous_soma_acc": 0.2502467772358995, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.3644418394603395, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6429523975968546 + }, + "split": "untouched_confirmation", + "wall_s": 2.3434809036552906, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603300, + "role_cosine_after_warmup": 0.9952875971794128, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603300, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603300, + "role_cosine_after_warmup": 0.9952875971794128, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603300, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603300, + "role_cosine_after_warmup": 0.9952875971794128, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603300, + "role_cosine_after_warmup": 0.9952875971794128, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m1.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m1.json new file mode 100644 index 0000000..564789e --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m1.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 1, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 33 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13533, + "acute_outcome_lesion": 13533, + "intact": 13533 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.9514689445495605, + "median": 1.8701362609863281, + "minimum": 1.649247646331787 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600033, + "targets": [ + 1.8291540622711182, + 1.8529463708400726, + 1.8701705932617188, + 1.8835307955741882, + 1.8972665786743164 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.8291540622711182, + 1.8529463708400726, + 1.8701705932617188, + 1.8835307955741882, + 1.8972665786743164 + ], + "trajectory_seeds": { + "1": 643000, + "2": 643001, + "3": 643002, + "4": 643003, + "5": 643004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590033 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 13426, + "cursor_scalar_observations": 7346, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3673, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.0, + 0.03125, + 0.046875, + 0.4375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9652366638183594, + "training_wall_s": 0.17380427196621895 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25070, + "cursor_scalar_observations": 12536, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6268, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06428223103284836, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.006917059421539307, + "training_wall_s": 0.22489523515105247 + }, + "intact": { + "cost": { + "active_state_episode_steps": 11579, + "cursor_scalar_observations": 6502, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3251, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5930096507072449, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.015625, + 0.09375, + 0.453125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 258, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9595192670822144, + "training_wall_s": 0.25554751977324486 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 11648, + "cursor_scalar_observations": 6526, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3263, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.650759220123291, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.015625, + 0.09375, + 0.46875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 259, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.1861511841416359 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 16265, + "cursor_scalar_observations": 8484, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4242, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.2131086140871048, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0625, + 0.15625, + 0.515625, + 0.6875, + 0.84375, + 0.921875, + 0.984375, + 0.984375, + 0.953125, + 0.953125 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 1776, + "final_success": 0.9609375, + "late_success": 0.9635416666666666, + "learning_gain": 0.9583333333333334, + "role_cosine_after_training": 0.9517040252685547, + "training_wall_s": 0.22179164364933968 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25070, + "cursor_scalar_observations": 12536, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6268, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06458590179681778, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9944956302642822, + "training_wall_s": 0.19708698615431786 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 950.1484375, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 1, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 33 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5291901149987767, + "acute_outcome_lesion_role_aligned_separation": 0.010341640793982254, + "causal_role_sign_inversion_index": 0.043850379234334005, + "challenge_episodes": 640, + "challenge_failure_count": 335, + "challenge_success_count": 305, + "challenge_success_fraction": 0.4765625, + "critic_contribution_value_prediction_corr": 0.9999999999998339, + "decoder_distance_residual_corr": 0.11542931750395755, + "mean_abs_raw_soma_corr": 0.9989128040453441, + "mean_abs_residual_soma_corr": 0.061892861945928836, + "mean_critic_expectedness_contribution": 0.29483485377787855, + "nonterminal_training_events": 10683, + "raw_minus_residual_abs_soma_corr": 0.9370199420994153, + "role_aligned_error_cv_corr": 0.3363631136924806, + "role_aligned_velocity_cv_corr": 0.998412627144863, + "surrounding_event_decoder_balanced_acc": 0.5424776378634735, + "target_success_fraction": { + "1.8291540622711182": 0.765625, + "1.8529463708400726": 0.671875, + "1.8701705932617188": 0.4140625, + "1.8835307955741882": 0.3046875, + "1.8972665786743164": 0.2265625 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.38808612932759556, + "terminal_previous_soma_outcome_balanced_acc": 0.7695130902862735, + "terminal_residual_minus_previous_soma_acc": 0.23048690971372654, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.3984277701215778, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6620495134523824 + }, + "split": "untouched_confirmation", + "wall_s": 2.366440463811159, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603301, + "role_cosine_after_warmup": 0.9976726770401001, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603301, + "role_cosine_after_warmup": 0.006917059421539307, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603301, + "role_cosine_after_warmup": 0.9976726770401001, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603301, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603301, + "role_cosine_after_warmup": 0.9976726770401001, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603301, + "role_cosine_after_warmup": 0.9976726770401001, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m2.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m2.json new file mode 100644 index 0000000..fb7f168 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m2.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 2, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 33 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12872, + "acute_outcome_lesion": 12872, + "intact": 12872 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.856842041015625, + "median": 1.768274188041687, + "minimum": 1.6334586143493652 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600033, + "targets": [ + 1.7394506454467773, + 1.7572676301002503, + 1.768316388130188, + 1.7820612132549285, + 1.8002171993255616 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7394506454467773, + 1.7572676301002503, + 1.768316388130188, + 1.7820612132549285, + 1.8002171993255616 + ], + "trajectory_seeds": { + "1": 643000, + "2": 643001, + "3": 643002, + "4": 643003, + "5": 643004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590033 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 11043, + "cursor_scalar_observations": 6264, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3132, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.015625, + 0.015625, + 0.046875, + 0.21875, + 0.625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 0.9755688905715942, + "training_wall_s": 0.17668503522872925 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.07305049896240234, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.09573888778686523, + "training_wall_s": 0.2136816382408142 + }, + "intact": { + "cost": { + "active_state_episode_steps": 8484, + "cursor_scalar_observations": 5134, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2567, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5640550851821899, + "daily_success": [ + 0.0, + 0.015625, + 0.03125, + 0.296875, + 0.984375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.015625, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.984375, + "role_cosine_after_training": 0.989530086517334, + "training_wall_s": 0.23928479850292206 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 8445, + "cursor_scalar_observations": 5116, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2558, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5714468359947205, + "daily_success": [ + 0.0, + 0.015625, + 0.03125, + 0.296875, + 0.984375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.015625, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.984375, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.17503183707594872 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 16738, + "cursor_scalar_observations": 8694, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4347, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.24022117257118225, + "daily_success": [ + 0.0, + 0.015625, + 0.015625, + 0.046875, + 0.140625, + 0.171875, + 0.3125, + 0.484375, + 0.84375, + 0.90625, + 0.9375, + 0.96875, + 1.0, + 0.984375 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 1505, + "final_success": 0.97265625, + "late_success": 0.984375, + "learning_gain": 0.9739583333333334, + "role_cosine_after_training": 0.9430809617042542, + "training_wall_s": 0.23192191496491432 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.07377980649471283, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9945967197418213, + "training_wall_s": 0.20052997022867203 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 937.1171875, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 2, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 33 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.6493772893772893, + "acute_outcome_lesion_role_aligned_separation": -0.0025771857552241983, + "causal_role_sign_inversion_index": 0.045717171038709786, + "challenge_episodes": 640, + "challenge_failure_count": 315, + "challenge_success_count": 325, + "challenge_success_fraction": 0.5078125, + "critic_contribution_value_prediction_corr": 0.9999999999993571, + "decoder_distance_residual_corr": 0.08121293789958617, + "mean_abs_raw_soma_corr": 0.9984306498463565, + "mean_abs_residual_soma_corr": 0.06973322873993029, + "mean_critic_expectedness_contribution": 0.392415459378294, + "nonterminal_training_events": 7588, + "raw_minus_residual_abs_soma_corr": 0.9286974211064263, + "role_aligned_error_cv_corr": 0.35704886333685665, + "role_aligned_velocity_cv_corr": 0.9981354393024301, + "surrounding_event_decoder_balanced_acc": 0.5386270492657449, + "target_success_fraction": { + "1.7394506454467773": 0.8203125, + "1.7572676301002503": 0.703125, + "1.768316388130188": 0.5546875, + "1.7820612132549285": 0.3359375, + "1.8002171993255616": 0.125 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.4066291092338222, + "terminal_previous_soma_outcome_balanced_acc": 0.7191452991452991, + "terminal_residual_minus_previous_soma_acc": 0.2808547008547009, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.404051923478598, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6410865759655735 + }, + "split": "untouched_confirmation", + "wall_s": 2.2367754094302654, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603302, + "role_cosine_after_warmup": 0.9930546879768372, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603302, + "role_cosine_after_warmup": -0.09573888778686523, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603302, + "role_cosine_after_warmup": 0.9930546879768372, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603302, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603302, + "role_cosine_after_warmup": 0.9930546879768372, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603302, + "role_cosine_after_warmup": 0.9930546879768372, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m3.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m3.json new file mode 100644 index 0000000..2236b0d --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m3.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 3, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 33 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13370, + "acute_outcome_lesion": 13370, + "intact": 13370 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8446906805038452, + "median": 1.7518749237060547, + "minimum": 1.6421949863433838 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600033, + "targets": [ + 1.7198402643203736, + 1.73607257604599, + 1.7518752217292786, + 1.7640233874320983, + 1.7799914598464965 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7198402643203736, + 1.73607257604599, + 1.7518752217292786, + 1.7640233874320983, + 1.7799914598464965 + ], + "trajectory_seeds": { + "1": 643000, + "2": 643001, + "3": 643002, + "4": 643003, + "5": 643004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590033 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 8748, + "cursor_scalar_observations": 5236, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2618, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.03125, + 0.0, + 0.015625, + 0.21875, + 0.890625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.015625, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.984375, + "role_cosine_after_training": 0.9938035011291504, + "training_wall_s": 0.1479373686015606 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06881073862314224, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7152, + "final_success": 0.00390625, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.006882439833134413, + "training_wall_s": 0.2206757515668869 + }, + "intact": { + "cost": { + "active_state_episode_steps": 8950, + "cursor_scalar_observations": 5306, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2653, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5748723745346069, + "daily_success": [ + 0.03125, + 0.0, + 0.015625, + 0.21875, + 0.78125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.015625, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.984375, + "role_cosine_after_training": 0.9944673776626587, + "training_wall_s": 0.239554300904274 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 9032, + "cursor_scalar_observations": 5334, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2667, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.559136152267456, + "daily_success": [ + 0.03125, + 0.0, + 0.015625, + 0.21875, + 0.75, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.015625, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.984375, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.16826795414090157 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 16209, + "cursor_scalar_observations": 8470, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4235, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.3062616288661957, + "daily_success": [ + 0.03125, + 0.0, + 0.0, + 0.078125, + 0.109375, + 0.28125, + 0.484375, + 0.71875, + 0.90625, + 0.859375, + 0.984375, + 0.984375, + 0.953125, + 0.984375 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 1312, + "final_success": 0.98046875, + "late_success": 0.9739583333333334, + "learning_gain": 0.9635416666666666, + "role_cosine_after_training": 0.9383077621459961, + "training_wall_s": 0.21105661988258362 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06911115348339081, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7152, + "final_success": 0.00390625, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9945859909057617, + "training_wall_s": 0.19601381942629814 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 940.3671875, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 3, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 33 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5815925806230474, + "acute_outcome_lesion_role_aligned_separation": -0.0076081239108812815, + "causal_role_sign_inversion_index": 0.044677411675222864, + "challenge_episodes": 640, + "challenge_failure_count": 337, + "challenge_success_count": 303, + "challenge_success_fraction": 0.4734375, + "critic_contribution_value_prediction_corr": 0.9999999999994839, + "decoder_distance_residual_corr": 0.1396429311416465, + "mean_abs_raw_soma_corr": 0.9989232246267428, + "mean_abs_residual_soma_corr": 0.06329090313685179, + "mean_critic_expectedness_contribution": 0.41784613863149483, + "nonterminal_training_events": 8054, + "raw_minus_residual_abs_soma_corr": 0.935632321489891, + "role_aligned_error_cv_corr": 0.36087604058876166, + "role_aligned_velocity_cv_corr": 0.997946887323742, + "surrounding_event_decoder_balanced_acc": 0.5529014979346336, + "target_success_fraction": { + "1.7198402643203736": 0.7734375, + "1.73607257604599": 0.6484375, + "1.7518752217292786": 0.484375, + "1.7640233874320983": 0.3359375, + "1.7799914598464965": 0.125 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.39892900193069464, + "terminal_previous_soma_outcome_balanced_acc": 0.7717581847205492, + "terminal_residual_minus_previous_soma_acc": 0.22824181527945075, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.39132087801981336, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6370708467349804 + }, + "split": "untouched_confirmation", + "wall_s": 2.2021420300006866, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603303, + "role_cosine_after_warmup": 0.9957466721534729, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603303, + "role_cosine_after_warmup": -0.006882439833134413, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603303, + "role_cosine_after_warmup": 0.9957466721534729, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603303, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603303, + "role_cosine_after_warmup": 0.9957466721534729, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603303, + "role_cosine_after_warmup": 0.9957466721534729, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m4.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m4.json new file mode 100644 index 0000000..1509972 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m4.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 4, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 33 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13107, + "acute_outcome_lesion": 13107, + "intact": 13107 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8632211685180664, + "median": 1.769853115081787, + "minimum": 1.6404733657836914 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600033, + "targets": [ + 1.7335654973983765, + 1.7542202532291413, + 1.7698944807052612, + 1.7847568273544312, + 1.7983635902404784 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7335654973983765, + 1.7542202532291413, + 1.7698944807052612, + 1.7847568273544312, + 1.7983635902404784 + ], + "trajectory_seeds": { + "1": 643000, + "2": 643001, + "3": 643002, + "4": 643003, + "5": 643004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590033 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 14061, + "cursor_scalar_observations": 7604, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3802, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.046875, + 0.234375, + 0.828125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9818060994148254, + "training_wall_s": 0.1677936390042305 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06266044825315475, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.01729092001914978, + "training_wall_s": 0.2129843905568123 + }, + "intact": { + "cost": { + "active_state_episode_steps": 11316, + "cursor_scalar_observations": 6398, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3199, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5313290953636169, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.171875, + 0.640625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9905637502670288, + "training_wall_s": 0.2484220303595066 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 11371, + "cursor_scalar_observations": 6420, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3210, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5086159110069275, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.171875, + 0.625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 261, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.18337474390864372 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 17342, + "cursor_scalar_observations": 8998, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4499, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.24839924275875092, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.03125, + 0.09375, + 0.375, + 0.53125, + 0.734375, + 0.890625, + 0.921875, + 0.96875, + 0.953125, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 984, + "final_success": 1.0, + "late_success": 0.9739583333333334, + "learning_gain": 0.9739583333333334, + "role_cosine_after_training": 0.9726309180259705, + "training_wall_s": 0.22761574760079384 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0628105103969574, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.994706928730011, + "training_wall_s": 0.1864093840122223 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 941.4609375, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 4, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 33 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5762179919631596, + "acute_outcome_lesion_role_aligned_separation": -0.006221165825558039, + "causal_role_sign_inversion_index": 0.04251334097034519, + "challenge_episodes": 640, + "challenge_failure_count": 331, + "challenge_success_count": 309, + "challenge_success_fraction": 0.4828125, + "critic_contribution_value_prediction_corr": 0.9999999999989748, + "decoder_distance_residual_corr": 0.12414756381253607, + "mean_abs_raw_soma_corr": 0.9990286360437013, + "mean_abs_residual_soma_corr": 0.05772289053791879, + "mean_critic_expectedness_contribution": 0.36466310032141347, + "nonterminal_training_events": 10420, + "raw_minus_residual_abs_soma_corr": 0.9413057455057825, + "role_aligned_error_cv_corr": 0.3607334381312698, + "role_aligned_velocity_cv_corr": 0.9990128501524438, + "surrounding_event_decoder_balanced_acc": 0.5435074166860925, + "target_success_fraction": { + "1.7335654973983765": 0.828125, + "1.7542202532291413": 0.609375, + "1.7698944807052612": 0.4375, + "1.7847568273544312": 0.3046875, + "1.7983635902404784": 0.234375 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.3876345319548794, + "terminal_previous_soma_outcome_balanced_acc": 0.710101780424134, + "terminal_residual_minus_previous_soma_acc": 0.28989821957586603, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.38141336612932136, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.638279412021174 + }, + "split": "untouched_confirmation", + "wall_s": 2.3145771585404873, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603304, + "role_cosine_after_warmup": 0.994428277015686, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603304, + "role_cosine_after_warmup": 0.01729092001914978, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603304, + "role_cosine_after_warmup": 0.994428277015686, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603304, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603304, + "role_cosine_after_warmup": 0.994428277015686, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603304, + "role_cosine_after_warmup": 0.994428277015686, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m0.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m0.json new file mode 100644 index 0000000..31b3a88 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m0.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 0, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 34 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13034, + "acute_outcome_lesion": 13034, + "intact": 13034 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.9955346584320068, + "median": 1.961836338043213, + "minimum": 1.7151234149932861 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600034, + "targets": [ + 1.9322781085968017, + 1.951997697353363, + 1.961840033531189, + 1.970857572555542, + 1.9781808376312255 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.9322781085968017, + 1.951997697353363, + 1.961840033531189, + 1.970857572555542, + 1.9781808376312255 + ], + "trajectory_seeds": { + "1": 644000, + "2": 644001, + "3": 644002, + "4": 644003, + "5": 644004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590034 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 20669, + "cursor_scalar_observations": 10540, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 5270, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.03125, + 0.078125, + 0.328125, + 0.875, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 288, + "final_success": 1.0, + "late_success": 0.9583333333333334, + "learning_gain": 0.9583333333333334, + "role_cosine_after_training": 0.9930513501167297, + "training_wall_s": 0.19851215556263924 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.060365356504917145, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.2130940966308117 + }, + "intact": { + "cost": { + "active_state_episode_steps": 13627, + "cursor_scalar_observations": 7398, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3699, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.6785129904747009, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.109375, + 0.53125, + 0.984375, + 0.984375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 268, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9545974135398865, + "training_wall_s": 0.2799079567193985 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 13276, + "cursor_scalar_observations": 7266, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3633, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.47762027382850647, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.109375, + 0.484375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 257, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.1935276947915554 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 19721, + "cursor_scalar_observations": 10112, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 5056, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.17650610208511353, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.09375, + 0.09375, + 0.203125, + 0.46875, + 0.546875, + 0.734375, + 0.84375, + 0.9375, + 0.984375 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 1701, + "final_success": 0.953125, + "late_success": 0.921875, + "learning_gain": 0.921875, + "role_cosine_after_training": 0.966079592704773, + "training_wall_s": 0.22892732918262482 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06099257990717888, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9968063235282898, + "training_wall_s": 0.18701686710119247 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 958.8671875, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 0, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 34 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.6127077223851418, + "acute_outcome_lesion_role_aligned_separation": 0.06268501173589083, + "causal_role_sign_inversion_index": 0.04465153226455211, + "challenge_episodes": 640, + "challenge_failure_count": 330, + "challenge_success_count": 310, + "challenge_success_fraction": 0.484375, + "critic_contribution_value_prediction_corr": 0.9999999999999575, + "decoder_distance_residual_corr": 0.13772538719193206, + "mean_abs_raw_soma_corr": 0.9987264362996662, + "mean_abs_residual_soma_corr": 0.059574479043578446, + "mean_critic_expectedness_contribution": 0.07823988392392697, + "nonterminal_training_events": 12731, + "raw_minus_residual_abs_soma_corr": 0.9391519572560877, + "role_aligned_error_cv_corr": 0.3486775000648828, + "role_aligned_velocity_cv_corr": 0.9951312693256131, + "surrounding_event_decoder_balanced_acc": 0.5444936098214749, + "target_success_fraction": { + "1.9322781085968017": 0.8203125, + "1.951997697353363": 0.6015625, + "1.961840033531189": 0.5, + "1.970857572555542": 0.296875, + "1.9781808376312255": 0.203125 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.42116100954239943, + "terminal_previous_soma_outcome_balanced_acc": 0.7004398826979472, + "terminal_residual_minus_previous_soma_acc": 0.2874389051808406, + "terminal_residual_outcome_balanced_acc": 0.9878787878787878, + "terminal_role_aligned_outcome_separation": 0.4838460212782903, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6464537692607304 + }, + "split": "untouched_confirmation", + "wall_s": 2.5098069943487644, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603400, + "role_cosine_after_warmup": 0.9966025352478027, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603400, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603400, + "role_cosine_after_warmup": 0.9966025352478027, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603400, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603400, + "role_cosine_after_warmup": 0.9966025352478027, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603400, + "role_cosine_after_warmup": 0.9966025352478027, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m1.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m1.json new file mode 100644 index 0000000..a6b860e --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m1.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 1, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 34 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12879, + "acute_outcome_lesion": 12879, + "intact": 12879 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8866384029388428, + "median": 1.793442726135254, + "minimum": 1.649482250213623 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600034, + "targets": [ + 1.7527262449264527, + 1.7756527423858643, + 1.7938211560249329, + 1.8106845617294312, + 1.8292778730392456 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7527262449264527, + 1.7756527423858643, + 1.7938211560249329, + 1.8106845617294312, + 1.8292778730392456 + ], + "trajectory_seeds": { + "1": 644000, + "2": 644001, + "3": 644002, + "4": 644003, + "5": 644004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590034 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 9297, + "cursor_scalar_observations": 5488, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2744, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0625, + 0.09375, + 0.671875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.020833333333333332, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9791666666666666, + "role_cosine_after_training": 0.9951645135879517, + "training_wall_s": 0.16294650733470917 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0671127513051033, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7158, + "final_success": 0.00390625, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.006917059421539307, + "training_wall_s": 0.2263321317732334 + }, + "intact": { + "cost": { + "active_state_episode_steps": 8464, + "cursor_scalar_observations": 5120, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2560, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5543029308319092, + "daily_success": [ + 0.0, + 0.0, + 0.078125, + 0.296875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.026041666666666668, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9739583333333334, + "role_cosine_after_training": 0.992432713508606, + "training_wall_s": 0.2567439265549183 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 8500, + "cursor_scalar_observations": 5122, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2561, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.565598726272583, + "daily_success": [ + 0.0, + 0.0, + 0.078125, + 0.296875, + 0.984375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.026041666666666668, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9739583333333334, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.1690628081560135 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 15372, + "cursor_scalar_observations": 8092, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4046, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.29054734110832214, + "daily_success": [ + 0.0, + 0.0, + 0.0625, + 0.015625, + 0.125, + 0.40625, + 0.65625, + 0.734375, + 0.8125, + 0.921875, + 1.0, + 1.0, + 1.0, + 0.96875 + ], + "early_success": 0.020833333333333332, + "evaluation_active_state_episode_steps": 1210, + "final_success": 0.99609375, + "late_success": 0.9895833333333334, + "learning_gain": 0.96875, + "role_cosine_after_training": 0.9424427151679993, + "training_wall_s": 0.2212667465209961 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06747626513242722, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7158, + "final_success": 0.00390625, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9968318343162537, + "training_wall_s": 0.19107519835233688 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 936.01171875, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 1, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 34 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5625, + "acute_outcome_lesion_role_aligned_separation": -0.005656169569584313, + "causal_role_sign_inversion_index": 0.04417944382511393, + "challenge_episodes": 640, + "challenge_failure_count": 320, + "challenge_success_count": 320, + "challenge_success_fraction": 0.5, + "critic_contribution_value_prediction_corr": 0.9999999999993817, + "decoder_distance_residual_corr": 0.10087124305513093, + "mean_abs_raw_soma_corr": 0.9988059352146047, + "mean_abs_residual_soma_corr": 0.06473228929829619, + "mean_critic_expectedness_contribution": 0.40755316659916485, + "nonterminal_training_events": 7568, + "raw_minus_residual_abs_soma_corr": 0.9340736459163085, + "role_aligned_error_cv_corr": 0.35574136239103776, + "role_aligned_velocity_cv_corr": 0.9980016006068788, + "surrounding_event_decoder_balanced_acc": 0.5432240964630742, + "target_success_fraction": { + "1.7527262449264527": 0.859375, + "1.7756527423858643": 0.625, + "1.7938211560249329": 0.484375, + "1.8106845617294312": 0.328125, + "1.8292778730392456": 0.203125 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.4244711195039915, + "terminal_previous_soma_outcome_balanced_acc": 0.7296875, + "terminal_residual_minus_previous_soma_acc": 0.27031249999999996, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.4188149499344072, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6422602382158411 + }, + "split": "untouched_confirmation", + "wall_s": 2.242401834577322, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603401, + "role_cosine_after_warmup": 0.9955573678016663, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603401, + "role_cosine_after_warmup": 0.006917059421539307, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603401, + "role_cosine_after_warmup": 0.9955573678016663, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603401, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603401, + "role_cosine_after_warmup": 0.9955573678016663, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603401, + "role_cosine_after_warmup": 0.9955573678016663, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m2.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m2.json new file mode 100644 index 0000000..9e30aa2 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m2.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 2, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 34 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13140, + "acute_outcome_lesion": 13140, + "intact": 13140 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.952148675918579, + "median": 1.8761571645736694, + "minimum": 1.6765412092208862 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600034, + "targets": [ + 1.8388256072998046, + 1.8584989547729491, + 1.8761998414993286, + 1.8894396007061005, + 1.90291907787323 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.8388256072998046, + 1.8584989547729491, + 1.8761998414993286, + 1.8894396007061005, + 1.90291907787323 + ], + "trajectory_seeds": { + "1": 644000, + "2": 644001, + "3": 644002, + "4": 644003, + "5": 644004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590034 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 9602, + "cursor_scalar_observations": 5634, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2817, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.03125, + 0.078125, + 0.578125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 0.9899946451187134, + "training_wall_s": 0.15460075438022614 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.07274568825960159, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.09573888778686523, + "training_wall_s": 0.2143809236586094 + }, + "intact": { + "cost": { + "active_state_episode_steps": 8569, + "cursor_scalar_observations": 5200, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2600, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5651165246963501, + "daily_success": [ + 0.0, + 0.0, + 0.0625, + 0.28125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.020833333333333332, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9791666666666666, + "role_cosine_after_training": 0.9955602288246155, + "training_wall_s": 0.2422812208533287 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 8561, + "cursor_scalar_observations": 5192, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2596, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5690587759017944, + "daily_success": [ + 0.0, + 0.0, + 0.0625, + 0.28125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.020833333333333332, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9791666666666666, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.1598205231130123 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 13833, + "cursor_scalar_observations": 7364, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3682, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.30188265442848206, + "daily_success": [ + 0.0, + 0.0, + 0.03125, + 0.09375, + 0.28125, + 0.40625, + 0.6875, + 0.890625, + 0.953125, + 0.984375, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 686, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 0.94648277759552, + "training_wall_s": 0.204206895083189 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0734877809882164, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9967483878135681, + "training_wall_s": 0.1902415044605732 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 937.78515625, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 2, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 34 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5647785948990769, + "acute_outcome_lesion_role_aligned_separation": 0.004273182819625576, + "causal_role_sign_inversion_index": 0.04451350704218453, + "challenge_episodes": 640, + "challenge_failure_count": 332, + "challenge_success_count": 308, + "challenge_success_fraction": 0.48125, + "critic_contribution_value_prediction_corr": 0.9999999999991698, + "decoder_distance_residual_corr": 0.10101998587950463, + "mean_abs_raw_soma_corr": 0.9985223766642548, + "mean_abs_residual_soma_corr": 0.06709978590606933, + "mean_critic_expectedness_contribution": 0.3358282675554531, + "nonterminal_training_events": 7673, + "raw_minus_residual_abs_soma_corr": 0.9314225907581855, + "role_aligned_error_cv_corr": 0.3426914856704604, + "role_aligned_velocity_cv_corr": 0.9987905394460401, + "surrounding_event_decoder_balanced_acc": 0.5344735932024204, + "target_success_fraction": { + "1.8388256072998046": 0.84375, + "1.8584989547729491": 0.6171875, + "1.8761998414993286": 0.5078125, + "1.8894396007061005": 0.21875, + "1.90291907787323": 0.21875 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.4025361835753929, + "terminal_previous_soma_outcome_balanced_acc": 0.7559458613675482, + "terminal_residual_minus_previous_soma_acc": 0.24405413863245184, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.40680936639501847, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6560990537755798 + }, + "split": "untouched_confirmation", + "wall_s": 2.171383861452341, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603402, + "role_cosine_after_warmup": 0.9935503005981445, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603402, + "role_cosine_after_warmup": -0.09573888778686523, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603402, + "role_cosine_after_warmup": 0.9935503005981445, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603402, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603402, + "role_cosine_after_warmup": 0.9935503005981445, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603402, + "role_cosine_after_warmup": 0.9935503005981445, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m3.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m3.json new file mode 100644 index 0000000..98a49b8 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m3.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 3, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 34 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13235, + "acute_outcome_lesion": 13235, + "intact": 13235 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.829050064086914, + "median": 1.7393901348114014, + "minimum": 1.6205710172653198 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600034, + "targets": [ + 1.7060932159423827, + 1.7256744980812073, + 1.7394548058509827, + 1.7526567339897157, + 1.769277238845825 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7060932159423827, + 1.7256744980812073, + 1.7394548058509827, + 1.7526567339897157, + 1.769277238845825 + ], + "trajectory_seeds": { + "1": 644000, + "2": 644001, + "3": 644002, + "4": 644003, + "5": 644004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590034 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 6762, + "cursor_scalar_observations": 4382, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2191, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.046875, + 0.328125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.125, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.875, + "role_cosine_after_training": 0.9929644465446472, + "training_wall_s": 0.15552960336208344 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06777670234441757, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.006882439833134413, + "training_wall_s": 0.21283961459994316 + }, + "intact": { + "cost": { + "active_state_episode_steps": 7007, + "cursor_scalar_observations": 4472, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2236, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5885887742042542, + "daily_success": [ + 0.0, + 0.046875, + 0.328125, + 0.984375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.125, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.875, + "role_cosine_after_training": 0.9924469590187073, + "training_wall_s": 0.2320079766213894 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 7023, + "cursor_scalar_observations": 4472, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2236, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5778533220291138, + "daily_success": [ + 0.0, + 0.046875, + 0.328125, + 0.984375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.125, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.875, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.16389469429850578 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 14202, + "cursor_scalar_observations": 7532, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3766, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.3136417865753174, + "daily_success": [ + 0.0, + 0.046875, + 0.109375, + 0.1875, + 0.421875, + 0.515625, + 0.734375, + 0.921875, + 0.921875, + 0.9375, + 0.953125, + 0.96875, + 0.984375, + 1.0 + ], + "early_success": 0.052083333333333336, + "evaluation_active_state_episode_steps": 1173, + "final_success": 0.99609375, + "late_success": 0.984375, + "learning_gain": 0.9322916666666666, + "role_cosine_after_training": 0.9585977792739868, + "training_wall_s": 0.22589751705527306 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06808338314294815, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9967212677001953, + "training_wall_s": 0.19004155322909355 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 930.40234375, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 3, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 34 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.564243927688957, + "acute_outcome_lesion_role_aligned_separation": 0.004005386507278497, + "causal_role_sign_inversion_index": 0.04602067993624194, + "challenge_episodes": 640, + "challenge_failure_count": 317, + "challenge_success_count": 323, + "challenge_success_fraction": 0.5046875, + "critic_contribution_value_prediction_corr": 0.9999999999988731, + "decoder_distance_residual_corr": 0.13883664983229876, + "mean_abs_raw_soma_corr": 0.9988093490303159, + "mean_abs_residual_soma_corr": 0.0689045272889682, + "mean_critic_expectedness_contribution": 0.43610042763651047, + "nonterminal_training_events": 6111, + "raw_minus_residual_abs_soma_corr": 0.9299048217413477, + "role_aligned_error_cv_corr": 0.3787894027978074, + "role_aligned_velocity_cv_corr": 0.997863820134255, + "surrounding_event_decoder_balanced_acc": 0.5458093195901826, + "target_success_fraction": { + "1.7060932159423827": 0.8125, + "1.7256744980812073": 0.640625, + "1.7394548058509827": 0.4453125, + "1.7526567339897157": 0.390625, + "1.769277238845825": 0.234375 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.4188848367647028, + "terminal_previous_soma_outcome_balanced_acc": 0.7118008418708676, + "terminal_residual_minus_previous_soma_acc": 0.28819915812913244, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.4228902232719813, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6190744173364476 + }, + "split": "untouched_confirmation", + "wall_s": 2.1345403268933296, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603403, + "role_cosine_after_warmup": 0.9959805607795715, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603403, + "role_cosine_after_warmup": -0.006882439833134413, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603403, + "role_cosine_after_warmup": 0.9959805607795715, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603403, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603403, + "role_cosine_after_warmup": 0.9959805607795715, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603403, + "role_cosine_after_warmup": 0.9959805607795715, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m4.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m4.json new file mode 100644 index 0000000..b199d25 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m4.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 4, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 34 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13290, + "acute_outcome_lesion": 13290, + "intact": 13290 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8710882663726807, + "median": 1.768843650817871, + "minimum": 1.5561155080795288 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600034, + "targets": [ + 1.7203734397888184, + 1.747604924440384, + 1.7688528299331665, + 1.7823114335536956, + 1.7989168167114258 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7203734397888184, + 1.747604924440384, + 1.7688528299331665, + 1.7823114335536956, + 1.7989168167114258 + ], + "trajectory_seeds": { + "1": 644000, + "2": 644001, + "3": 644002, + "4": 644003, + "5": 644004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590034 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 16565, + "cursor_scalar_observations": 8714, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4357, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.046875, + 0.53125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9851133227348328, + "training_wall_s": 0.18903668224811554 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06220182776451111, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.01729092001914978, + "training_wall_s": 0.230977650731802 + }, + "intact": { + "cost": { + "active_state_episode_steps": 12769, + "cursor_scalar_observations": 7026, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3513, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.4841112494468689, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.046875, + 0.109375, + 0.734375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 259, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9479692578315735, + "training_wall_s": 0.2660930007696152 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 12778, + "cursor_scalar_observations": 7028, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3514, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5170386433601379, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.046875, + 0.109375, + 0.734375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 261, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.18882036581635475 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 17123, + "cursor_scalar_observations": 8908, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4454, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.20128095149993896, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.046875, + 0.0625, + 0.3125, + 0.546875, + 0.71875, + 0.90625, + 0.921875, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 980, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.959757387638092, + "training_wall_s": 0.22321565821766853 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.062158528715372086, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9967688918113708, + "training_wall_s": 0.20211265608668327 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 950.234375, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 4, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 34 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5528505015904086, + "acute_outcome_lesion_role_aligned_separation": -0.005125152666244803, + "causal_role_sign_inversion_index": 0.03966625170379978, + "challenge_episodes": 640, + "challenge_failure_count": 335, + "challenge_success_count": 305, + "challenge_success_fraction": 0.4765625, + "critic_contribution_value_prediction_corr": 0.9999999999992801, + "decoder_distance_residual_corr": 0.1162612006043953, + "mean_abs_raw_soma_corr": 0.9991388391535155, + "mean_abs_residual_soma_corr": 0.04957770864352022, + "mean_critic_expectedness_contribution": 0.3888222075031653, + "nonterminal_training_events": 11873, + "raw_minus_residual_abs_soma_corr": 0.9495611305099954, + "role_aligned_error_cv_corr": 0.3557026952559581, + "role_aligned_velocity_cv_corr": 0.9964927020726014, + "surrounding_event_decoder_balanced_acc": 0.5423245682096336, + "target_success_fraction": { + "1.7203734397888184": 0.75, + "1.747604924440384": 0.5390625, + "1.7688528299331665": 0.4765625, + "1.7823114335536956": 0.40625, + "1.7989168167114258": 0.2109375 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.46559291415555737, + "terminal_previous_soma_outcome_balanced_acc": 0.7856373868363102, + "terminal_residual_minus_previous_soma_acc": 0.21436261316368976, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.46046776148931257, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6407900068166433 + }, + "split": "untouched_confirmation", + "wall_s": 2.4391065910458565, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603404, + "role_cosine_after_warmup": 0.9952877759933472, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603404, + "role_cosine_after_warmup": 0.01729092001914978, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603404, + "role_cosine_after_warmup": 0.9952877759933472, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603404, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603404, + "role_cosine_after_warmup": 0.9952877759933472, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603404, + "role_cosine_after_warmup": 0.9952877759933472, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m0.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m0.json new file mode 100644 index 0000000..50c50cb --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m0.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 0, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 35 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12850, + "acute_outcome_lesion": 12850, + "intact": 12850 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.9849567413330078, + "median": 1.91834557056427, + "minimum": 1.6202419996261597 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600035, + "targets": [ + 1.8735879898071288, + 1.8998247623443603, + 1.9183531999588013, + 1.9367210805416106, + 1.95015230178833 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.8735879898071288, + 1.8998247623443603, + 1.9183531999588013, + 1.9367210805416106, + 1.95015230178833 + ], + "trajectory_seeds": { + "1": 645000, + "2": 645001, + "3": 645002, + "4": 645003, + "5": 645004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590035 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9956165552139282, + "training_wall_s": 0.19498155638575554 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06231851130723953, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21101489663124084 + }, + "intact": { + "cost": { + "active_state_episode_steps": 14900, + "cursor_scalar_observations": 7982, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3991, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5462889671325684, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.046875, + 0.125, + 0.53125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 276, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.9648510217666626, + "training_wall_s": 0.28073331341147423 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 14892, + "cursor_scalar_observations": 7988, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3994, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5053926706314087, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0625, + 0.140625, + 0.53125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 260, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.1843544878065586 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 19869, + "cursor_scalar_observations": 10178, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 5089, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.17359036207199097, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.046875, + 0.046875, + 0.078125, + 0.359375, + 0.609375, + 0.765625, + 0.921875, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 992, + "final_success": 0.9921875, + "late_success": 0.9739583333333334, + "learning_gain": 0.9739583333333334, + "role_cosine_after_training": 0.975403904914856, + "training_wall_s": 0.20823364332318306 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06312094628810883, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9956166744232178, + "training_wall_s": 0.18489129841327667 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 957.69140625, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 0, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 35 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.6287701203313018, + "acute_outcome_lesion_role_aligned_separation": 0.028752198912096877, + "causal_role_sign_inversion_index": 0.03884229397369869, + "challenge_episodes": 640, + "challenge_failure_count": 324, + "challenge_success_count": 316, + "challenge_success_fraction": 0.49375, + "critic_contribution_value_prediction_corr": 0.9999999999995777, + "decoder_distance_residual_corr": 0.1377614810313545, + "mean_abs_raw_soma_corr": 0.9991352034552479, + "mean_abs_residual_soma_corr": 0.05274924660445603, + "mean_critic_expectedness_contribution": 0.15648079138243814, + "nonterminal_training_events": 14004, + "raw_minus_residual_abs_soma_corr": 0.9463859568507919, + "role_aligned_error_cv_corr": 0.34843011120216355, + "role_aligned_velocity_cv_corr": 0.997885961402315, + "surrounding_event_decoder_balanced_acc": 0.5553637651318272, + "target_success_fraction": { + "1.8735879898071288": 0.859375, + "1.8998247623443603": 0.6484375, + "1.9183531999588013": 0.484375, + "1.9367210805416106": 0.2890625, + "1.95015230178833": 0.1875 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.38941009406111604, + "terminal_previous_soma_outcome_balanced_acc": 0.7147796530707924, + "terminal_residual_minus_previous_soma_acc": 0.2836771370526644, + "terminal_residual_outcome_balanced_acc": 0.9984567901234568, + "terminal_role_aligned_outcome_separation": 0.41816229297321295, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6494558502001515 + }, + "split": "untouched_confirmation", + "wall_s": 2.4860168024897575, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603500, + "role_cosine_after_warmup": 0.9947201013565063, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603500, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603500, + "role_cosine_after_warmup": 0.9947201013565063, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603500, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603500, + "role_cosine_after_warmup": 0.9947201013565063, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603500, + "role_cosine_after_warmup": 0.9947201013565063, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m1.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m1.json new file mode 100644 index 0000000..37b60f3 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m1.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 1, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 35 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13253, + "acute_outcome_lesion": 13253, + "intact": 13253 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8374848365783691, + "median": 1.7371206283569336, + "minimum": 1.5806999206542969 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600035, + "targets": [ + 1.7027083873748778, + 1.7213188648223876, + 1.7372562289237976, + 1.7519998967647552, + 1.7660725116729736 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.7027083873748778, + 1.7213188648223876, + 1.7372562289237976, + 1.7519998967647552, + 1.7660725116729736 + ], + "trajectory_seeds": { + "1": 645000, + "2": 645001, + "3": 645002, + "4": 645003, + "5": 645004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590035 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 9453, + "cursor_scalar_observations": 5560, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2780, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.046875, + 0.125, + 0.578125, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.015625, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.984375, + "role_cosine_after_training": 0.9872337579727173, + "training_wall_s": 0.17131321877241135 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06651832908391953, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.006917059421539307, + "training_wall_s": 0.23586415871977806 + }, + "intact": { + "cost": { + "active_state_episode_steps": 9384, + "cursor_scalar_observations": 5510, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2755, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5480557680130005, + "daily_success": [ + 0.0, + 0.0, + 0.046875, + 0.1875, + 0.625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.015625, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.984375, + "role_cosine_after_training": 0.9893856644630432, + "training_wall_s": 0.2562956213951111 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 9148, + "cursor_scalar_observations": 5396, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2698, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5525104999542236, + "daily_success": [ + 0.0, + 0.0, + 0.046875, + 0.203125, + 0.6875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.015625, + "evaluation_active_state_episode_steps": 257, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.984375, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.17988957464694977 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 15270, + "cursor_scalar_observations": 8048, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4024, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.25545960664749146, + "daily_success": [ + 0.0, + 0.0, + 0.03125, + 0.09375, + 0.15625, + 0.3125, + 0.484375, + 0.84375, + 0.90625, + 0.9375, + 0.96875, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 724, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 0.9530673027038574, + "training_wall_s": 0.22465555742383003 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06670410931110382, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9956286549568176, + "training_wall_s": 0.20815648883581161 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 939.0859375, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 1, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 35 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.6134285490992928, + "acute_outcome_lesion_role_aligned_separation": -0.005400645623502676, + "causal_role_sign_inversion_index": 0.042955956857375924, + "challenge_episodes": 640, + "challenge_failure_count": 314, + "challenge_success_count": 326, + "challenge_success_fraction": 0.509375, + "critic_contribution_value_prediction_corr": 0.999999999998843, + "decoder_distance_residual_corr": 0.10877909177817344, + "mean_abs_raw_soma_corr": 0.9988967447967028, + "mean_abs_residual_soma_corr": 0.06510835310816333, + "mean_critic_expectedness_contribution": 0.35947227975731383, + "nonterminal_training_events": 8488, + "raw_minus_residual_abs_soma_corr": 0.9337883916885394, + "role_aligned_error_cv_corr": 0.3520701485325682, + "role_aligned_velocity_cv_corr": 0.9985743694658887, + "surrounding_event_decoder_balanced_acc": 0.5353584643992392, + "target_success_fraction": { + "1.7027083873748778": 0.8203125, + "1.7213188648223876": 0.6796875, + "1.7372562289237976": 0.5078125, + "1.7519998967647552": 0.3515625, + "1.7660725116729736": 0.1875 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.40415192077101025, + "terminal_previous_soma_outcome_balanced_acc": 0.7487397913328904, + "terminal_residual_minus_previous_soma_acc": 0.25126020866710963, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.3987512751475076, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6465042209333205 + }, + "split": "untouched_confirmation", + "wall_s": 2.352902326732874, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603501, + "role_cosine_after_warmup": 0.9961993098258972, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603501, + "role_cosine_after_warmup": 0.006917059421539307, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603501, + "role_cosine_after_warmup": 0.9961993098258972, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603501, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603501, + "role_cosine_after_warmup": 0.9961993098258972, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603501, + "role_cosine_after_warmup": 0.9961993098258972, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m2.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m2.json new file mode 100644 index 0000000..9c09022 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m2.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 2, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 35 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12835, + "acute_outcome_lesion": 12835, + "intact": 12835 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.9992141723632812, + "median": 1.9736332893371582, + "minimum": 0.8970615863800049 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600035, + "targets": [ + 1.9343899726867675, + 1.9582673728466033, + 1.9737194776535034, + 1.9839000523090362, + 1.989602518081665 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.9343899726867675, + 1.9582673728466033, + 1.9737194776535034, + 1.9839000523090362, + 1.989602518081665 + ], + "trajectory_seeds": { + "1": 645000, + "2": 645001, + "3": 645002, + "4": 645003, + "5": 645004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590035 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9954671859741211, + "training_wall_s": 0.19426430389285088 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.07122565805912018, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.09573888778686523, + "training_wall_s": 0.2128269337117672 + }, + "intact": { + "cost": { + "active_state_episode_steps": 11730, + "cursor_scalar_observations": 6532, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3266, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.7352330088615417, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.03125, + 0.109375, + 0.484375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 497, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 0.8839390277862549, + "training_wall_s": 0.2566399984061718 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 11453, + "cursor_scalar_observations": 6448, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3224, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5193716287612915, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.03125, + 0.109375, + 0.515625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 1.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.16714980080723763 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 17376, + "cursor_scalar_observations": 9030, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4515, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.2626461088657379, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.078125, + 0.078125, + 0.3125, + 0.421875, + 0.734375, + 0.828125, + 0.953125, + 0.96875, + 0.984375, + 1.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 910, + "final_success": 0.99609375, + "late_success": 0.984375, + "learning_gain": 0.984375, + "role_cosine_after_training": 0.9630982875823975, + "training_wall_s": 0.208146370947361 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.07205408066511154, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9954672455787659, + "training_wall_s": 0.1851862482726574 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 949.8125, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 2, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 35 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.7482786573038647, + "acute_outcome_lesion_role_aligned_separation": 0.24923514105752506, + "causal_role_sign_inversion_index": 0.046814937940840094, + "challenge_episodes": 640, + "challenge_failure_count": 317, + "challenge_success_count": 323, + "challenge_success_fraction": 0.5046875, + "critic_contribution_value_prediction_corr": 0.9999999999999933, + "decoder_distance_residual_corr": 0.11793383977089922, + "mean_abs_raw_soma_corr": 0.9982828737398639, + "mean_abs_residual_soma_corr": 0.06927720885884571, + "mean_critic_expectedness_contribution": -0.058464501718956285, + "nonterminal_training_events": 10834, + "raw_minus_residual_abs_soma_corr": 0.9290056648810182, + "role_aligned_error_cv_corr": 0.3649085954540223, + "role_aligned_velocity_cv_corr": 0.9908071769186144, + "surrounding_event_decoder_balanced_acc": 0.5451387139481143, + "target_success_fraction": { + "1.9343899726867675": 0.8359375, + "1.9582673728466033": 0.703125, + "1.9737194776535034": 0.4921875, + "1.9839000523090362": 0.28125, + "1.989602518081665": 0.2109375 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.39172167405837177, + "terminal_previous_soma_outcome_balanced_acc": 0.7412419060268969, + "terminal_residual_minus_previous_soma_acc": 0.2350987879794122, + "terminal_residual_outcome_balanced_acc": 0.9763406940063091, + "terminal_role_aligned_outcome_separation": 0.6409568151158969, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6258985814645921 + }, + "split": "untouched_confirmation", + "wall_s": 2.3440728820860386, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603502, + "role_cosine_after_warmup": 0.9930879473686218, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603502, + "role_cosine_after_warmup": -0.09573888778686523, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603502, + "role_cosine_after_warmup": 0.9930879473686218, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603502, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603502, + "role_cosine_after_warmup": 0.9930879473686218, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603502, + "role_cosine_after_warmup": 0.9930879473686218, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m3.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m3.json new file mode 100644 index 0000000..fff2d4a --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m3.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 3, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 35 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 12705, + "acute_outcome_lesion": 12705, + "intact": 12705 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.9576371908187866, + "median": 1.8692450523376465, + "minimum": 1.6967694759368896 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600035, + "targets": [ + 1.8278280258178712, + 1.8505628049373626, + 1.8692976236343384, + 1.8833139359951019, + 1.9005751132965087 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.8278280258178712, + 1.8505628049373626, + 1.8692976236343384, + 1.8833139359951019, + 1.9005751132965087 + ], + "trajectory_seeds": { + "1": 645000, + "2": 645001, + "3": 645002, + "4": 645003, + "5": 645004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590035 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 9378, + "cursor_scalar_observations": 5510, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2755, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.140625, + 0.625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9946816563606262, + "training_wall_s": 0.15270386263728142 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25086, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06561318784952164, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.006882439833134413, + "training_wall_s": 0.21236185729503632 + }, + "intact": { + "cost": { + "active_state_episode_steps": 9382, + "cursor_scalar_observations": 5506, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2753, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.6080682873725891, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.1875, + 0.640625, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 258, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9178956151008606, + "training_wall_s": 0.2465122453868389 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 8341, + "cursor_scalar_observations": 5080, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 2540, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5769819617271423, + "daily_success": [ + 0.0, + 0.0, + 0.03125, + 0.375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 257, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.16422712057828903 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 14766, + "cursor_scalar_observations": 7776, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3888, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.2823747992515564, + "daily_success": [ + 0.0, + 0.0, + 0.015625, + 0.109375, + 0.1875, + 0.34375, + 0.625, + 0.796875, + 0.921875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 886, + "final_success": 0.9921875, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9015244245529175, + "training_wall_s": 0.20319201797246933 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25086, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06602432578802109, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9955136775970459, + "training_wall_s": 0.18580254912376404 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 936.640625, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 3, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 35 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5139942772878641, + "acute_outcome_lesion_role_aligned_separation": 0.0034805028770052426, + "causal_role_sign_inversion_index": 0.041752666348701825, + "challenge_episodes": 640, + "challenge_failure_count": 321, + "challenge_success_count": 319, + "challenge_success_fraction": 0.4984375, + "critic_contribution_value_prediction_corr": 0.9999999999996196, + "decoder_distance_residual_corr": 0.13551523054590273, + "mean_abs_raw_soma_corr": 0.9990584279550836, + "mean_abs_residual_soma_corr": 0.06293747217568228, + "mean_critic_expectedness_contribution": 0.35577087429511256, + "nonterminal_training_events": 8486, + "raw_minus_residual_abs_soma_corr": 0.9361209557794014, + "role_aligned_error_cv_corr": 0.3579806654398201, + "role_aligned_velocity_cv_corr": 0.9983302984418154, + "surrounding_event_decoder_balanced_acc": 0.547604641332401, + "target_success_fraction": { + "1.8278280258178712": 0.859375, + "1.8505628049373626": 0.640625, + "1.8692976236343384": 0.4140625, + "1.8833139359951019": 0.34375, + "1.9005751132965087": 0.234375 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.4249036044697145, + "terminal_previous_soma_outcome_balanced_acc": 0.6970966513344856, + "terminal_residual_minus_previous_soma_acc": 0.3029033486655144, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.4283841073467197, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6403496330019953 + }, + "split": "untouched_confirmation", + "wall_s": 2.1912553794682026, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603503, + "role_cosine_after_warmup": 0.9930452108383179, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603503, + "role_cosine_after_warmup": -0.006882439833134413, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603503, + "role_cosine_after_warmup": 0.9930452108383179, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603503, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603503, + "role_cosine_after_warmup": 0.9930452108383179, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603503, + "role_cosine_after_warmup": 0.9930452108383179, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m4.json b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m4.json new file mode 100644 index 0000000..d6ba057 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m4.json @@ -0,0 +1,468 @@ +{ + "args": { + "model_seed": 4, + "outdir": "results/bci_v2_calibrated_confirmation", + "r1_gate": "results/bci_v2_calibrated_dev_gate.json", + "task_seed": 35 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 13049, + "acute_outcome_lesion": 13049, + "intact": 13049 + }, + "calibration": { + "active_state_episode_steps": 14336, + "episodes": 512, + "maximum_cursor_summary": { + "maximum": 1.8010891675949097, + "median": 1.702605962753296, + "minimum": 1.5472567081451416 + }, + "quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "seed": 600035, + "targets": [ + 1.6710152864456176, + 1.6885900616645813, + 1.7027654647827148, + 1.7147585809230805, + 1.7336405038833618 + ], + "uses_outcome_labels": false + }, + "episodes_per_target": 128, + "maximum_steps_per_episode": 28, + "selection_over_evaluation_outcomes": false, + "targets": [ + 1.6710152864456176, + 1.6885900616645813, + 1.7027654647827148, + 1.7147585809230805, + 1.7336405038833618 + ], + "trajectory_seeds": { + "1": 645000, + "2": 645001, + "3": 645002, + "4": 645003, + "5": 645004 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 590035 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 13508, + "cursor_scalar_observations": 7382, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3691, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.015625, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0625, + 0.375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 259, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9938409924507141, + "training_wall_s": 0.1641547866165638 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06417792290449142, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.01729092001914978, + "training_wall_s": 0.21302392333745956 + }, + "intact": { + "cost": { + "active_state_episode_steps": 12185, + "cursor_scalar_observations": 6762, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3381, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.507308840751648, + "daily_success": [ + 0.015625, + 0.0, + 0.0, + 0.0, + 0.0625, + 0.3125, + 0.921875, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 256, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9947916666666666, + "role_cosine_after_training": 0.9901720881462097, + "training_wall_s": 0.24989206343889236 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 10820, + "cursor_scalar_observations": 6162, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 3081, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.5275433659553528, + "daily_success": [ + 0.015625, + 0.0, + 0.015625, + 0.03125, + 0.171875, + 0.859375, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "early_success": 0.010416666666666666, + "evaluation_active_state_episode_steps": 257, + "final_success": 1.0, + "late_success": 1.0, + "learning_gain": 0.9895833333333334, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.16610369086265564 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 17599, + "cursor_scalar_observations": 9114, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 4557, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.22303137183189392, + "daily_success": [ + 0.015625, + 0.0, + 0.0, + 0.0, + 0.03125, + 0.0625, + 0.234375, + 0.484375, + 0.78125, + 0.796875, + 0.96875, + 0.96875, + 1.0, + 1.0 + ], + "early_success": 0.005208333333333333, + "evaluation_active_state_episode_steps": 968, + "final_success": 1.0, + "late_success": 0.9895833333333334, + "learning_gain": 0.984375, + "role_cosine_after_training": 0.9803339838981628, + "training_wall_s": 0.20659048482775688 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.06426402926445007, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9953718185424805, + "training_wall_s": 0.18532931059598923 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 1.0 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 947.6875, + "protocol": { + "calibration_uses_outcome_labels": false, + "confirmation_grid_size": 30, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "model_seed": 4, + "name": "oral_b_v2_calibrated_recovery_confirmation_v1", + "no_further_selection": true, + "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate_sha256": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "split": "untouched_confirmation", + "training_task_seed": 35 + }, + "provenance": { + "git_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "common_runner": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "development_runner": true, + "failed_target_gate": true, + "failed_v2_gate": true, + "old_r2_gate": true, + "protocol": true, + "r1_gate": true, + "recovery_confirmation_analyzer": true, + "recovery_development_analyzer": true, + "recovery_metrics": true, + "recovery_runner": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 4, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.6371308016877637, + "acute_outcome_lesion_role_aligned_separation": -0.0026569205266886997, + "causal_role_sign_inversion_index": 0.04128954239023773, + "challenge_episodes": 640, + "challenge_failure_count": 316, + "challenge_success_count": 324, + "challenge_success_fraction": 0.50625, + "critic_contribution_value_prediction_corr": 0.999999999998732, + "decoder_distance_residual_corr": 0.13045368055712556, + "mean_abs_raw_soma_corr": 0.9990883804588234, + "mean_abs_residual_soma_corr": 0.061279949528038946, + "mean_critic_expectedness_contribution": 0.3526887600338273, + "nonterminal_training_events": 11289, + "raw_minus_residual_abs_soma_corr": 0.9378084309307845, + "role_aligned_error_cv_corr": 0.3649381696115974, + "role_aligned_velocity_cv_corr": 0.9982809405487985, + "surrounding_event_decoder_balanced_acc": 0.545167143013386, + "target_success_fraction": { + "1.6710152864456176": 0.7734375, + "1.6885900616645813": 0.6796875, + "1.7027654647827148": 0.515625, + "1.7147585809230805": 0.328125, + "1.7336405038833618": 0.234375 + }, + "terminal_outcome_separation_drop_under_acute_lesion": 0.40951872557604313, + "terminal_previous_soma_outcome_balanced_acc": 0.7504102203469292, + "terminal_residual_minus_previous_soma_acc": 0.24958977965307083, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.40686180504935443, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.633342770937201 + }, + "split": "untouched_confirmation", + "wall_s": 2.298864256590605, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603504, + "role_cosine_after_warmup": 0.9950474500656128, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603504, + "role_cosine_after_warmup": 0.01729092001914978, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603504, + "role_cosine_after_warmup": 0.9950474500656128, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603504, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603504, + "role_cosine_after_warmup": 0.9950474500656128, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 603504, + "role_cosine_after_warmup": 0.9950474500656128, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_calibrated_confirmation_gate.json b/results/bci_v2_calibrated_confirmation_gate.json new file mode 100644 index 0000000..90e54e5 --- /dev/null +++ b/results/bci_v2_calibrated_confirmation_gate.json @@ -0,0 +1,385 @@ +{ + "all_prior_failures_preserved": true, + "calibration_quantiles": [ + 0.2, + 0.35, + 0.5, + 0.65, + 0.8 + ], + "checks": { + "all_records_finite_paired_and_cost_audited": true, + "innovation_and_network_prediction": { + "decoder_corr_lower_bound_nonnegative": true, + "mean_decoder_corr_at_least_0p05": true, + "mean_raw_residual_corr_gap_at_least_0p20": true, + "mean_residual_soma_corr_at_most_0p10": true, + "mean_surrounding_accuracy_at_least_0p52": true, + "mean_velocity_advantage_at_least_0p05": true, + "positive_sign_in_at_least_25_of_30": true, + "positive_sign_in_every_task_cluster": true, + "raw_residual_corr_gap_lower_bound_at_least_0p15": true, + "residual_soma_corr_upper_bound_at_most_0p12": true, + "surrounding_accuracy_lower_bound_at_least_0p50": true, + "velocity_advantage_lower_bound_nonnegative": true + }, + "learning_and_plasticity": { + "every_role_cosine_at_least_0p70": true, + "every_task_mean_final_at_least_0p60": true, + "fixed_role_gap_lower_bound_at_least_0p10": true, + "intact_final_lower_bound_at_least_0p60": true, + "learning_gain_lower_bound_at_least_0p05": true, + "mean_fixed_role_gap_at_least_0p20": true, + "mean_intact_final_at_least_0p70": true, + "mean_learning_gain_at_least_0p10": true, + "mean_oracle_deficit_at_most_0p10": true, + "mean_role_cosine_at_least_0p80": true, + "oracle_deficit_upper_bound_at_most_0p20": true, + "plasticity_half_margin_lower_bound_nonnegative": true + }, + "target_ladder_outcome_surprise": { + "critic_expectedness_lower_bound_at_least_0p02": true, + "every_challenge_fraction_between_0p10_and_0p90": true, + "every_critic_value_corr_at_least_0p95": true, + "mean_critic_expectedness_at_least_0p05": true, + "mean_outcome_lesion_drop_at_least_0p20": true, + "mean_terminal_accuracy_at_least_0p80": true, + "mean_terminal_separation_at_least_0p20": true, + "outcome_lesion_drop_lower_bound_at_least_0p15": true, + "terminal_accuracy_lower_bound_at_least_0p75": true, + "terminal_separation_lower_bound_at_least_0p15": true + } + }, + "complete_grid": true, + "fixed_config": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "velocity_reward_scale": 1.0 + }, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c", + "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4", + "development_runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18", + "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1", + "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8", + "r1_gate": "907648edaea040b0d7e6fbf74f96b9c5ff8fc19c0dc1063b054829d5f2aa9fbc", + "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b", + "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2", + "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa", + "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a", + "runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "metrics": { + "challenge_fraction": { + "by_task_seed": [ + 0.493125, + 0.5209375, + 0.514375, + 0.4853125, + 0.489375, + 0.5025 + ], + "mean": 0.5009375, + "one_sided_95pct_lower": 0.4891770555188329, + "one_sided_95pct_upper": 0.5126979444811671 + }, + "critic_expectedness": { + "by_task_seed": [ + 0.3133184602399429, + 0.32414884320577275, + 0.35313381879463884, + 0.35924697778590636, + 0.32930879064364416, + 0.2331896407499471 + ], + "mean": 0.3187244219033087, + "one_sided_95pct_lower": 0.2813635909438583, + "one_sided_95pct_upper": 0.35608525286275905 + }, + "critic_training_gap": { + "by_task_seed": [ + 0.19921875, + 0.13984375, + 0.19921875, + 0.0, + 0.0, + 0.4 + ], + "mean": 0.15638020833333333, + "one_sided_95pct_lower": 0.03314670866020128, + "one_sided_95pct_upper": 0.27961370800646534 + }, + "critic_value_corr": { + "by_task_seed": [ + 0.9999999999993657, + 0.9999999999992956, + 0.9999999999990888, + 0.9999999999994579, + 0.9999999999993324, + 0.9999999999993532 + ], + "mean": 0.9999999999993157, + "one_sided_95pct_lower": 0.9999999999992141, + "one_sided_95pct_upper": 0.9999999999994172 + }, + "decoder_corr": { + "by_task_seed": [ + 0.12762928350655744, + 0.11863155855659262, + 0.11915750059779097, + 0.11494479466762646, + 0.11894289331265234, + 0.1260886647366911 + ], + "mean": 0.1208991158963185, + "one_sided_95pct_lower": 0.11687435649320946, + "one_sided_95pct_upper": 0.12492387529942753 + }, + "fixed_role_gap": { + "by_task_seed": [ + 1.0, + 1.0, + 1.0, + 0.99921875, + 0.99921875, + 1.0 + ], + "mean": 0.9997395833333333, + "one_sided_95pct_lower": 0.9994077009138491, + "one_sided_95pct_upper": 1.0000714657528176 + }, + "intact_final": { + "by_task_seed": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "mean": 1.0, + "one_sided_95pct_lower": 1.0, + "one_sided_95pct_upper": 1.0 + }, + "intact_gain": { + "by_task_seed": [ + 0.9885416666666667, + 0.9958333333333333, + 0.990625, + 0.9927083333333333, + 0.965625, + 0.9947916666666666 + ], + "mean": 0.9880208333333333, + "one_sided_95pct_lower": 0.978732077409237, + "one_sided_95pct_upper": 0.9973095892574297 + }, + "oracle_deficit": { + "by_task_seed": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "mean": 0.0, + "one_sided_95pct_lower": 0.0, + "one_sided_95pct_upper": 0.0 + }, + "outcome_lesion_drop": { + "by_task_seed": [ + 0.39257537041898616, + 0.3907302442715473, + 0.3942709986960553, + 0.392041901233741, + 0.4265292127084088, + 0.4039412037872511 + ], + "mean": 0.40001482185266496, + "one_sided_95pct_lower": 0.3886365638440749, + "one_sided_95pct_upper": 0.411393079861255 + }, + "outcome_training_gap": { + "by_task_seed": [ + 0.00625, + 0.0140625, + 0.00703125, + 0.0203125, + 0.0109375, + 0.00390625 + ], + "mean": 0.010416666666666666, + "one_sided_95pct_lower": 0.005443964824466358, + "one_sided_95pct_upper": 0.015389368508866973 + }, + "plasticity_half_margin": { + "by_task_seed": [ + 0.4942708333333333, + 0.49895833333333334, + 0.4953125, + 0.49635416666666665, + 0.4828125, + 0.4973958333333333 + ], + "mean": 0.49418402777777776, + "one_sided_95pct_lower": 0.48940971207433226, + "one_sided_95pct_upper": 0.49895834348122325 + }, + "raw_residual_corr_gap": { + "by_task_seed": [ + 0.9343675823071695, + 0.9346327542596294, + 0.9351443461556045, + 0.9371585744517263, + 0.9368228292363849, + 0.936621880026107 + ], + "mean": 0.935791327739437, + "one_sided_95pct_lower": 0.9347897922928049, + "one_sided_95pct_upper": 0.936792863186069 + }, + "residual_soma_corr": { + "by_task_seed": [ + 0.06455947485925856, + 0.06426190360381064, + 0.06372971991387162, + 0.06172674968637641, + 0.061977758036086475, + 0.06227044605503726 + ], + "mean": 0.06308767535907349, + "one_sided_95pct_lower": 0.062066201125334404, + "one_sided_95pct_upper": 0.06410914959281258 + }, + "role_cosine": { + "by_task_seed": [ + 0.9775842428207397, + 0.984451150894165, + 0.9840954661369323, + 0.9847705602645874, + 0.9766013145446777, + 0.9492486834526062 + ], + "mean": 0.9761252363522848, + "one_sided_95pct_lower": 0.9648921083365222, + "one_sided_95pct_upper": 0.9873583643680474 + }, + "sign_inversion": { + "by_task_seed": [ + 0.0417874453823463, + 0.04286598907057731, + 0.04230713275613263, + 0.04317331085758009, + 0.04380628295437846, + 0.042331079502170854 + ], + "mean": 0.04271187342053094, + "one_sided_95pct_lower": 0.042118910888504044, + "one_sided_95pct_upper": 0.04330483595255784 + }, + "surrounding_accuracy": { + "by_task_seed": [ + 0.5426530288446466, + 0.5446535266803437, + 0.5430696812667446, + 0.5439424567783643, + 0.5420650374573571, + 0.5457265455649936 + ], + "mean": 0.5436850460987417, + "one_sided_95pct_lower": 0.5425667337998379, + "one_sided_95pct_upper": 0.5448033583976455 + }, + "terminal_accuracy": { + "by_task_seed": [ + 0.9987403100775194, + 0.998997277676951, + 0.9996865203761756, + 1.0, + 0.9975757575757576, + 0.9949594968259532 + ], + "mean": 0.9983265604220595, + "one_sided_95pct_lower": 0.9968020433207192, + "one_sided_95pct_upper": 0.9998510775233997 + }, + "terminal_separation": { + "by_task_seed": [ + 0.4020353802157061, + 0.3913365663582604, + 0.39170622720181636, + 0.38793115544193, + 0.43856566447380196, + 0.4586232591265383 + ], + "mean": 0.4116997088030089, + "one_sided_95pct_lower": 0.38730641450983544, + "one_sided_95pct_upper": 0.4360930030961823 + }, + "velocity_advantage": { + "by_task_seed": [ + 0.6370962829746177, + 0.6361040647361883, + 0.6326329660812359, + 0.644287749154193, + 0.6409354970810485, + 0.639110211307452 + ], + "mean": 0.6383611285557892, + "one_sided_95pct_lower": 0.6350333620000043, + "one_sided_95pct_upper": 0.6416888951115741 + } + }, + "new_oral_a_v2_protocol_may_be_frozen": true, + "old_oral_a_gate_remains_closed": true, + "oral_b_v2_outcome_surprise_established": true, + "positive_sign_count": 30, + "protocol": "oral_b_v2_calibrated_recovery_confirmation_v1", + "review_score_after": 8, + "review_score_before": 7, + "score_change_rule": "only a complete untouched calibrated R2 pass establishes role-vectorized TD outcome surprise; prior failures and the old oral-A gate remain unchanged", + "source_commit": "70e180c5ef5f78679f2e163ed3ee30873ee523bf", + "source_sha256": { + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m0.json": "2909c54021e9d451ef54ae5de6b521478119b9bffaca06f58f711b24431c4d06", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m1.json": "775b835a8d48f6d8dea2cb0078446552fb7cfab2d3e437aa9631c311cd6f40e2", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m2.json": "0f95bdeb1a156ef4a97a59748453dfa1f872b490e68e13df926e37057a42eb36", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m3.json": "fec845763b8402c8abf021000041d3b0790b5b50979ef381c1e2b016a9c5fee1", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t30_m4.json": "110fa43b9b0fcbcef89ed8a2332e871777c116690024ed2fa4bedea155fbb246", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m0.json": "1f1c1fa49dd56743081d6420193b56ecba5f11cb7b2a40300f26e666bc4fb821", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m1.json": "4efde69c884057c73d68a712625f5d940fb152e0ad7f5e301ffbecf9c9b8379a", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m2.json": "10e16869c8b23007b840bec177378abe18c26f3fbabaa9dc09d112d4655b140e", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m3.json": "a8be20d34915f95f7c278c4a4a3c2eaaf775d280b42614dfad8256cd877fe789", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t31_m4.json": "38e58907ad43ab06361a2946935855ea9d059a0ffd51d3cbd0ad596e1b923271", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m0.json": "0254fc9d1945d11c5f178b9267ecbf8fe7f7ac2af8099b84095b4e931e1bb719", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m1.json": "359c7a46d1f3bdde2e9394b5f7dceea01511b4a1714688976be3d69b4312a370", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m2.json": "b2e1228d511fffd9753aacfd2b8f534652612d348900d44584f3c8f7e9ce7c05", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m3.json": "e34d2a5e285184c6e60d088b67d8658d6dc78c62b2fc42c06acdb78f55bd6803", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t32_m4.json": "033dd12ed1e2efacc07ae14297bb0cc8fc74dc9114cd57431de9794bee7c7695", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m0.json": "fba80b70c40169efcb60144e8315b5837184463601c86f3e3bda21361b3318ed", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m1.json": "be952f8068e07612ef469dab0c495788d5b5b9dbc1c198b84ef5f9a17be3f62e", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m2.json": "1d144e27aff463aacbc2e418bb5567016690477ba6f1e9e0e84f241cc8f24105", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m3.json": "f81a7942503b1768c5819aff76e4525c1f0425c70b5a9f1ba397a30b0d5b31f2", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t33_m4.json": "42cc8ca59a3239c351bd3436bd8662e08c706f0861fce07266958e8e55d6dcd4", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m0.json": "b798f59f61ff25c1f6d8e06b35727492a901ee87d8798700d2d7ff111cff0ecb", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m1.json": "f5d4d84beeefd5c2e46187ccb61ef4432b0193c4c7f053c40252740a939ee651", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m2.json": "7b8a5e144f71c5c6476be92612b8924dcdf1ba1d8f3083ce80618099b756adc1", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m3.json": "c75f0c42a28ee1b764a7181e5fb7f449fecf84a3a63f69bf0663ed29f4dd1846", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t34_m4.json": "eb3de9a7edb8d3283a8762c1202310775d634af9a8ea5a25f90f314412f8f615", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m0.json": "c026bcf06003c4129fe2d444f30331187b872ea2d28d55e41e5752cf2e3acb15", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m1.json": "42452e471e10d9b17916f906b6b205632aa1f1de8335b1ffe962e855c53ee4e1", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m2.json": "37ecff67bc1f7899b95f81c5587eb826602c1053b6777d76076bb891742161df", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m3.json": "e53a422eff2bb2961ea396f071ea040a71d921e348a35bcb2d7e54d0952b3c23", + "results/bci_v2_calibrated_confirmation/bci_v2_calibrated_confirm_t35_m4.json": "4d9d35cb6d9bc30d240c85a0a5887c7d983e81acdb4a94b613bc633a11ea3178" + }, + "status": "passed" +} |
