summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorYurenHao0426 <Blackhao0426@gmail.com>2026-07-23 08:13:02 -0500
committerYurenHao0426 <Blackhao0426@gmail.com>2026-07-23 08:13:02 -0500
commit70e180c5ef5f78679f2e163ed3ee30873ee523bf (patch)
treeb3234a7a2311aafefd8e945c55e31551289d1058
parentd289a46293e0b422c195746ad8b310d4b50ded23 (diff)
results: pass calibrated oral-B-v2 development gate
-rw-r--r--results/bci_v2_calibrated_dev/bci_v2_calibrated_t26_m0.json461
-rw-r--r--results/bci_v2_calibrated_dev/bci_v2_calibrated_t27_m0.json461
-rw-r--r--results/bci_v2_calibrated_dev/bci_v2_calibrated_t28_m0.json461
-rw-r--r--results/bci_v2_calibrated_dev_gate.json148
4 files changed, 1531 insertions, 0 deletions
diff --git a/results/bci_v2_calibrated_dev/bci_v2_calibrated_t26_m0.json b/results/bci_v2_calibrated_dev/bci_v2_calibrated_t26_m0.json
new file mode 100644
index 0000000..b2fc4f3
--- /dev/null
+++ b/results/bci_v2_calibrated_dev/bci_v2_calibrated_t26_m0.json
@@ -0,0 +1,461 @@
+{
+ "args": {
+ "model_seed": 0,
+ "outdir": "results/bci_v2_calibrated_dev",
+ "task_seed": 26
+ },
+ "assays": {
+ "challenge": {
+ "active_state_episode_steps_by_mode": {
+ "acute_critic_lesion": 12882,
+ "acute_outcome_lesion": 12882,
+ "intact": 12882
+ },
+ "calibration": {
+ "active_state_episode_steps": 14336,
+ "episodes": 512,
+ "maximum_cursor_summary": {
+ "maximum": 1.9664530754089355,
+ "median": 1.9004390239715576,
+ "minimum": 1.6664398908615112
+ },
+ "quantiles": [
+ 0.2,
+ 0.35,
+ 0.5,
+ 0.65,
+ 0.8
+ ],
+ "seed": 550026,
+ "targets": [
+ 1.859693741798401,
+ 1.8864485919475555,
+ 1.9004749059677124,
+ 1.9156511783599854,
+ 1.929451322555542
+ ],
+ "uses_outcome_labels": false
+ },
+ "episodes_per_target": 128,
+ "maximum_steps_per_episode": 28,
+ "selection_over_evaluation_outcomes": false,
+ "targets": [
+ 1.859693741798401,
+ 1.8864485919475555,
+ 1.9004749059677124,
+ 1.9156511783599854,
+ 1.929451322555542
+ ],
+ "trajectory_seeds": {
+ "1": 586000,
+ "2": 586001,
+ "3": 586002,
+ "4": 586003,
+ "5": 586004
+ }
+ },
+ "performance_evaluation_episodes": 256,
+ "performance_evaluation_seed": 540026
+ },
+ "conditions": {
+ "critic_training_lesion": {
+ "cost": {
+ "active_state_episode_steps": 25088,
+ "cursor_scalar_observations": 12544,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 6272,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.0,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 7168,
+ "final_success": 0.0,
+ "late_success": 0.0,
+ "learning_gain": 0.0,
+ "role_cosine_after_training": 0.9942126274108887,
+ "training_wall_s": 0.21116646006703377
+ },
+ "fixed_role": {
+ "cost": {
+ "active_state_episode_steps": 25088,
+ "cursor_scalar_observations": 12544,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 6272,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.06177661940455437,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 7168,
+ "final_success": 0.0,
+ "late_success": 0.0,
+ "learning_gain": 0.0,
+ "role_cosine_after_training": -0.1393236517906189,
+ "training_wall_s": 0.21275918185710907
+ },
+ "intact": {
+ "cost": {
+ "active_state_episode_steps": 16204,
+ "cursor_scalar_observations": 8554,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 4277,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.5207661390304565,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.015625,
+ 0.203125,
+ 0.6875,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 263,
+ "final_success": 1.0,
+ "late_success": 1.0,
+ "learning_gain": 1.0,
+ "role_cosine_after_training": 0.9405146241188049,
+ "training_wall_s": 0.2710344232618809
+ },
+ "oracle_role": {
+ "cost": {
+ "active_state_episode_steps": 15936,
+ "cursor_scalar_observations": 8436,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 4218,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.45849862694740295,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.03125,
+ 0.265625,
+ 0.796875,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 267,
+ "final_success": 1.0,
+ "late_success": 1.0,
+ "learning_gain": 1.0,
+ "role_cosine_after_training": 1.0000001192092896,
+ "training_wall_s": 0.2009308896958828
+ },
+ "outcome_training_lesion": {
+ "cost": {
+ "active_state_episode_steps": 19102,
+ "cursor_scalar_observations": 9788,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 4894,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.18997597694396973,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.015625,
+ 0.15625,
+ 0.25,
+ 0.671875,
+ 0.8125,
+ 0.96875,
+ 1.0,
+ 1.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 641,
+ "final_success": 1.0,
+ "late_success": 0.9895833333333334,
+ "learning_gain": 0.9895833333333334,
+ "role_cosine_after_training": 0.9652196764945984,
+ "training_wall_s": 0.22571271285414696
+ },
+ "plasticity_lesion": {
+ "cost": {
+ "active_state_episode_steps": 25088,
+ "cursor_scalar_observations": 12544,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 6272,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.06284985691308975,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 7168,
+ "final_success": 0.0,
+ "late_success": 0.0,
+ "learning_gain": 0.0,
+ "role_cosine_after_training": 0.9942126870155334,
+ "training_wall_s": 0.20297767594456673
+ }
+ },
+ "config": {
+ "context_ar": 0.8,
+ "context_dim": 16,
+ "coupling_scale": 1.0,
+ "critic_eta": 0.03,
+ "days": 14,
+ "eligibility_decay": 0.8,
+ "episodes_per_day": 64,
+ "feedback": "performance_velocity",
+ "forward_eta": 0.1,
+ "gamma": 0.8,
+ "inertia": 0.65,
+ "kappa": 0.0,
+ "n_background": 30,
+ "n_minus": 5,
+ "n_plus": 5,
+ "perturb_every": 4,
+ "perturb_sigma": 0.03,
+ "predictor_eta": 0.2,
+ "process_noise": 0.12,
+ "steps_per_episode": 28,
+ "target": 0.8,
+ "terminal_reward": 1.0,
+ "vectorizer_eta": 0.03,
+ "velocity_reward_scale": 1.0
+ },
+ "finite": true,
+ "hardware": {
+ "device": "cpu",
+ "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35",
+ "threads": 1,
+ "torch_version": "2.10.0+cu128"
+ },
+ "peak_rss_mib": 962.34375,
+ "protocol": {
+ "calibration_uses_outcome_labels": false,
+ "confirmation_seeds_touched": false,
+ "failed_target_gate_sha256": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1",
+ "fixed_config": {
+ "critic_eta": 0.03,
+ "forward_eta": 0.1,
+ "gamma": 0.8,
+ "velocity_reward_scale": 1.0
+ },
+ "hyperparameter_selection": false,
+ "name": "oral_b_v2_calibrated_recovery_development_v1",
+ "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8",
+ "selection_split": "development_validation"
+ },
+ "provenance": {
+ "git_commit": "d289a46293e0b422c195746ad8b310d4b50ded23",
+ "git_tracked_dirty": false,
+ "input_sha256": {
+ "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc",
+ "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5",
+ "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c",
+ "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1",
+ "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef",
+ "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4",
+ "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1",
+ "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd",
+ "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597",
+ "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8",
+ "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b",
+ "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2",
+ "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa",
+ "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a",
+ "runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18",
+ "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae",
+ "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee"
+ },
+ "tracked_inputs": {
+ "base_dynamics": true,
+ "common_runner": true,
+ "confirmation_analyzer": true,
+ "confirmation_runner": true,
+ "d4_gate": true,
+ "development_analyzer": true,
+ "failed_target_gate": true,
+ "failed_v2_gate": true,
+ "old_r2_gate": true,
+ "protocol": true,
+ "recovery_confirmation_analyzer": true,
+ "recovery_development_analyzer": true,
+ "recovery_metrics": true,
+ "recovery_runner": true,
+ "runner": true,
+ "v2_dynamics": true,
+ "v2_metrics": true
+ }
+ },
+ "schema_version": 4,
+ "signatures": {
+ "acute_outcome_lesion_outcome_balanced_acc": 0.5002197802197802,
+ "acute_outcome_lesion_role_aligned_separation": 0.0005471409294360297,
+ "causal_role_sign_inversion_index": 0.03878189930912869,
+ "challenge_episodes": 640,
+ "challenge_failure_count": 315,
+ "challenge_success_count": 325,
+ "challenge_success_fraction": 0.5078125,
+ "critic_contribution_value_prediction_corr": 0.9999999999995302,
+ "decoder_distance_residual_corr": 0.11469715057185319,
+ "mean_abs_raw_soma_corr": 0.9992015754075411,
+ "mean_abs_residual_soma_corr": 0.0563586791161443,
+ "mean_critic_expectedness_contribution": 0.24124016763074801,
+ "nonterminal_training_events": 15308,
+ "raw_minus_residual_abs_soma_corr": 0.9428428962913967,
+ "role_aligned_error_cv_corr": 0.32966432771863374,
+ "role_aligned_velocity_cv_corr": 0.9988614599944285,
+ "surrounding_event_decoder_balanced_acc": 0.5539961311201976,
+ "target_success_fraction": {
+ "1.859693741798401": 0.828125,
+ "1.8864485919475555": 0.71875,
+ "1.9004749059677124": 0.4453125,
+ "1.9156511783599854": 0.328125,
+ "1.929451322555542": 0.21875
+ },
+ "terminal_outcome_separation_drop_under_acute_lesion": 0.3956960119743165,
+ "terminal_previous_soma_outcome_balanced_acc": 0.7272771672771673,
+ "terminal_residual_minus_previous_soma_acc": 0.26796092796092796,
+ "terminal_residual_outcome_balanced_acc": 0.9952380952380953,
+ "terminal_role_aligned_outcome_separation": 0.3962431529037525,
+ "terminal_training_events": 896,
+ "velocity_minus_error_abs_cv_corr": 0.6691971322757948
+ },
+ "split": "development_validation",
+ "wall_s": 2.583215858787298,
+ "warmup": {
+ "critic_training_lesion": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602600,
+ "role_cosine_after_warmup": 0.9944602251052856,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": true
+ },
+ "fixed_role": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602600,
+ "role_cosine_after_warmup": -0.1393236517906189,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": false
+ },
+ "intact": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602600,
+ "role_cosine_after_warmup": 0.9944602251052856,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": true
+ },
+ "oracle_role": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602600,
+ "role_cosine_after_warmup": 1.0000001192092896,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": false
+ },
+ "outcome_training_lesion": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602600,
+ "role_cosine_after_warmup": 0.9944602251052856,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": true
+ },
+ "plasticity_lesion": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602600,
+ "role_cosine_after_warmup": 0.9944602251052856,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": true
+ }
+ }
+}
diff --git a/results/bci_v2_calibrated_dev/bci_v2_calibrated_t27_m0.json b/results/bci_v2_calibrated_dev/bci_v2_calibrated_t27_m0.json
new file mode 100644
index 0000000..f1e11a3
--- /dev/null
+++ b/results/bci_v2_calibrated_dev/bci_v2_calibrated_t27_m0.json
@@ -0,0 +1,461 @@
+{
+ "args": {
+ "model_seed": 0,
+ "outdir": "results/bci_v2_calibrated_dev",
+ "task_seed": 27
+ },
+ "assays": {
+ "challenge": {
+ "active_state_episode_steps_by_mode": {
+ "acute_critic_lesion": 13244,
+ "acute_outcome_lesion": 13244,
+ "intact": 13244
+ },
+ "calibration": {
+ "active_state_episode_steps": 14336,
+ "episodes": 512,
+ "maximum_cursor_summary": {
+ "maximum": 1.8732032775878906,
+ "median": 1.749955177307129,
+ "minimum": 1.5971075296401978
+ },
+ "quantiles": [
+ 0.2,
+ 0.35,
+ 0.5,
+ 0.65,
+ 0.8
+ ],
+ "seed": 550027,
+ "targets": [
+ 1.7139501571655273,
+ 1.7322039246559142,
+ 1.750077724456787,
+ 1.7649245738983155,
+ 1.783210277557373
+ ],
+ "uses_outcome_labels": false
+ },
+ "episodes_per_target": 128,
+ "maximum_steps_per_episode": 28,
+ "selection_over_evaluation_outcomes": false,
+ "targets": [
+ 1.7139501571655273,
+ 1.7322039246559142,
+ 1.750077724456787,
+ 1.7649245738983155,
+ 1.783210277557373
+ ],
+ "trajectory_seeds": {
+ "1": 587000,
+ "2": 587001,
+ "3": 587002,
+ "4": 587003,
+ "5": 587004
+ }
+ },
+ "performance_evaluation_episodes": 256,
+ "performance_evaluation_seed": 540027
+ },
+ "conditions": {
+ "critic_training_lesion": {
+ "cost": {
+ "active_state_episode_steps": 14474,
+ "cursor_scalar_observations": 7798,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 3899,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.0,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.03125,
+ 0.015625,
+ 0.1875,
+ 0.609375,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 256,
+ "final_success": 1.0,
+ "late_success": 1.0,
+ "learning_gain": 1.0,
+ "role_cosine_after_training": 0.9887208938598633,
+ "training_wall_s": 0.16654988750815392
+ },
+ "fixed_role": {
+ "cost": {
+ "active_state_episode_steps": 25088,
+ "cursor_scalar_observations": 12544,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 6272,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.06176371872425079,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 7168,
+ "final_success": 0.0,
+ "late_success": 0.0,
+ "learning_gain": 0.0,
+ "role_cosine_after_training": -0.1393236517906189,
+ "training_wall_s": 0.21138394996523857
+ },
+ "intact": {
+ "cost": {
+ "active_state_episode_steps": 11369,
+ "cursor_scalar_observations": 6412,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 3206,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.508084774017334,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.109375,
+ 0.671875,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 256,
+ "final_success": 1.0,
+ "late_success": 1.0,
+ "learning_gain": 1.0,
+ "role_cosine_after_training": 0.9776318073272705,
+ "training_wall_s": 0.24492322281002998
+ },
+ "oracle_role": {
+ "cost": {
+ "active_state_episode_steps": 11256,
+ "cursor_scalar_observations": 6362,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 3181,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.520680844783783,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.125,
+ 0.6875,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 256,
+ "final_success": 1.0,
+ "late_success": 1.0,
+ "learning_gain": 1.0,
+ "role_cosine_after_training": 1.0000001192092896,
+ "training_wall_s": 0.16702108830213547
+ },
+ "outcome_training_lesion": {
+ "cost": {
+ "active_state_episode_steps": 17895,
+ "cursor_scalar_observations": 9242,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 4621,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.22657130658626556,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.03125,
+ 0.078125,
+ 0.265625,
+ 0.390625,
+ 0.71875,
+ 0.859375,
+ 0.875,
+ 0.96875,
+ 0.9375,
+ 1.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 1375,
+ "final_success": 0.984375,
+ "late_success": 0.96875,
+ "learning_gain": 0.96875,
+ "role_cosine_after_training": 0.9808546304702759,
+ "training_wall_s": 0.2095155566930771
+ },
+ "plasticity_lesion": {
+ "cost": {
+ "active_state_episode_steps": 25088,
+ "cursor_scalar_observations": 12544,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 6272,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.06274955719709396,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 7168,
+ "final_success": 0.0,
+ "late_success": 0.0,
+ "learning_gain": 0.0,
+ "role_cosine_after_training": 0.9958968758583069,
+ "training_wall_s": 0.18501129373908043
+ }
+ },
+ "config": {
+ "context_ar": 0.8,
+ "context_dim": 16,
+ "coupling_scale": 1.0,
+ "critic_eta": 0.03,
+ "days": 14,
+ "eligibility_decay": 0.8,
+ "episodes_per_day": 64,
+ "feedback": "performance_velocity",
+ "forward_eta": 0.1,
+ "gamma": 0.8,
+ "inertia": 0.65,
+ "kappa": 0.0,
+ "n_background": 30,
+ "n_minus": 5,
+ "n_plus": 5,
+ "perturb_every": 4,
+ "perturb_sigma": 0.03,
+ "predictor_eta": 0.2,
+ "process_noise": 0.12,
+ "steps_per_episode": 28,
+ "target": 0.8,
+ "terminal_reward": 1.0,
+ "vectorizer_eta": 0.03,
+ "velocity_reward_scale": 1.0
+ },
+ "finite": true,
+ "hardware": {
+ "device": "cpu",
+ "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35",
+ "threads": 1,
+ "torch_version": "2.10.0+cu128"
+ },
+ "peak_rss_mib": 947.375,
+ "protocol": {
+ "calibration_uses_outcome_labels": false,
+ "confirmation_seeds_touched": false,
+ "failed_target_gate_sha256": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1",
+ "fixed_config": {
+ "critic_eta": 0.03,
+ "forward_eta": 0.1,
+ "gamma": 0.8,
+ "velocity_reward_scale": 1.0
+ },
+ "hyperparameter_selection": false,
+ "name": "oral_b_v2_calibrated_recovery_development_v1",
+ "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8",
+ "selection_split": "development_validation"
+ },
+ "provenance": {
+ "git_commit": "d289a46293e0b422c195746ad8b310d4b50ded23",
+ "git_tracked_dirty": false,
+ "input_sha256": {
+ "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc",
+ "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5",
+ "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c",
+ "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1",
+ "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef",
+ "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4",
+ "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1",
+ "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd",
+ "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597",
+ "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8",
+ "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b",
+ "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2",
+ "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa",
+ "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a",
+ "runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18",
+ "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae",
+ "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee"
+ },
+ "tracked_inputs": {
+ "base_dynamics": true,
+ "common_runner": true,
+ "confirmation_analyzer": true,
+ "confirmation_runner": true,
+ "d4_gate": true,
+ "development_analyzer": true,
+ "failed_target_gate": true,
+ "failed_v2_gate": true,
+ "old_r2_gate": true,
+ "protocol": true,
+ "recovery_confirmation_analyzer": true,
+ "recovery_development_analyzer": true,
+ "recovery_metrics": true,
+ "recovery_runner": true,
+ "runner": true,
+ "v2_dynamics": true,
+ "v2_metrics": true
+ }
+ },
+ "schema_version": 4,
+ "signatures": {
+ "acute_outcome_lesion_outcome_balanced_acc": 0.5622654784240151,
+ "acute_outcome_lesion_role_aligned_separation": -0.004976110356813113,
+ "causal_role_sign_inversion_index": 0.03865690938282479,
+ "challenge_episodes": 640,
+ "challenge_failure_count": 328,
+ "challenge_success_count": 312,
+ "challenge_success_fraction": 0.4875,
+ "critic_contribution_value_prediction_corr": 0.9999999999991853,
+ "decoder_distance_residual_corr": 0.08504896381928725,
+ "mean_abs_raw_soma_corr": 0.9990959548954148,
+ "mean_abs_residual_soma_corr": 0.05868808106050728,
+ "mean_critic_expectedness_contribution": 0.35806743977971284,
+ "nonterminal_training_events": 10473,
+ "raw_minus_residual_abs_soma_corr": 0.9404078738349075,
+ "role_aligned_error_cv_corr": 0.39857193059373297,
+ "role_aligned_velocity_cv_corr": 0.9978555331143931,
+ "surrounding_event_decoder_balanced_acc": 0.541416057442577,
+ "target_success_fraction": {
+ "1.7139501571655273": 0.796875,
+ "1.7322039246559142": 0.6640625,
+ "1.750077724456787": 0.390625,
+ "1.7649245738983155": 0.3828125,
+ "1.783210277557373": 0.203125
+ },
+ "terminal_outcome_separation_drop_under_acute_lesion": 0.40359597715821127,
+ "terminal_previous_soma_outcome_balanced_acc": 0.7347951844903065,
+ "terminal_residual_minus_previous_soma_acc": 0.26520481550969355,
+ "terminal_residual_outcome_balanced_acc": 1.0,
+ "terminal_role_aligned_outcome_separation": 0.39861986680139816,
+ "terminal_training_events": 896,
+ "velocity_minus_error_abs_cv_corr": 0.5992836025206602
+ },
+ "split": "development_validation",
+ "wall_s": 2.259972095489502,
+ "warmup": {
+ "critic_training_lesion": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602700,
+ "role_cosine_after_warmup": 0.9947453737258911,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": true
+ },
+ "fixed_role": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602700,
+ "role_cosine_after_warmup": -0.1393236517906189,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": false
+ },
+ "intact": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602700,
+ "role_cosine_after_warmup": 0.9947453737258911,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": true
+ },
+ "oracle_role": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602700,
+ "role_cosine_after_warmup": 1.0000001192092896,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": false
+ },
+ "outcome_training_lesion": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602700,
+ "role_cosine_after_warmup": 0.9947453737258911,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": true
+ },
+ "plasticity_lesion": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602700,
+ "role_cosine_after_warmup": 0.9947453737258911,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": true
+ }
+ }
+}
diff --git a/results/bci_v2_calibrated_dev/bci_v2_calibrated_t28_m0.json b/results/bci_v2_calibrated_dev/bci_v2_calibrated_t28_m0.json
new file mode 100644
index 0000000..3880956
--- /dev/null
+++ b/results/bci_v2_calibrated_dev/bci_v2_calibrated_t28_m0.json
@@ -0,0 +1,461 @@
+{
+ "args": {
+ "model_seed": 0,
+ "outdir": "results/bci_v2_calibrated_dev",
+ "task_seed": 28
+ },
+ "assays": {
+ "challenge": {
+ "active_state_episode_steps_by_mode": {
+ "acute_critic_lesion": 12957,
+ "acute_outcome_lesion": 12957,
+ "intact": 12957
+ },
+ "calibration": {
+ "active_state_episode_steps": 14336,
+ "episodes": 512,
+ "maximum_cursor_summary": {
+ "maximum": 1.9806733131408691,
+ "median": 1.9220820665359497,
+ "minimum": 1.6861028671264648
+ },
+ "quantiles": [
+ 0.2,
+ 0.35,
+ 0.5,
+ 0.65,
+ 0.8
+ ],
+ "seed": 550028,
+ "targets": [
+ 1.8872586727142333,
+ 1.9072148621082305,
+ 1.9221965074539185,
+ 1.9328129887580872,
+ 1.9453917741775513
+ ],
+ "uses_outcome_labels": false
+ },
+ "episodes_per_target": 128,
+ "maximum_steps_per_episode": 28,
+ "selection_over_evaluation_outcomes": false,
+ "targets": [
+ 1.8872586727142333,
+ 1.9072148621082305,
+ 1.9221965074539185,
+ 1.9328129887580872,
+ 1.9453917741775513
+ ],
+ "trajectory_seeds": {
+ "1": 588000,
+ "2": 588001,
+ "3": 588002,
+ "4": 588003,
+ "5": 588004
+ }
+ },
+ "performance_evaluation_episodes": 256,
+ "performance_evaluation_seed": 540028
+ },
+ "conditions": {
+ "critic_training_lesion": {
+ "cost": {
+ "active_state_episode_steps": 17384,
+ "cursor_scalar_observations": 9062,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 4531,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.0,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.015625,
+ 0.015625,
+ 0.0625,
+ 0.296875,
+ 0.859375,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 261,
+ "final_success": 1.0,
+ "late_success": 1.0,
+ "learning_gain": 1.0,
+ "role_cosine_after_training": 0.9773973226547241,
+ "training_wall_s": 0.1755989007651806
+ },
+ "fixed_role": {
+ "cost": {
+ "active_state_episode_steps": 25088,
+ "cursor_scalar_observations": 12544,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 6272,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.06155245006084442,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 7168,
+ "final_success": 0.0,
+ "late_success": 0.0,
+ "learning_gain": 0.0,
+ "role_cosine_after_training": -0.1393236517906189,
+ "training_wall_s": 0.21381374821066856
+ },
+ "intact": {
+ "cost": {
+ "active_state_episode_steps": 14224,
+ "cursor_scalar_observations": 7682,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 3841,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.533806562423706,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.03125,
+ 0.03125,
+ 0.203125,
+ 0.890625,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 260,
+ "final_success": 1.0,
+ "late_success": 1.0,
+ "learning_gain": 1.0,
+ "role_cosine_after_training": 0.9761794805526733,
+ "training_wall_s": 0.258306160569191
+ },
+ "oracle_role": {
+ "cost": {
+ "active_state_episode_steps": 14083,
+ "cursor_scalar_observations": 7614,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 3807,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.4723714292049408,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.046875,
+ 0.0625,
+ 0.234375,
+ 0.890625,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0,
+ 1.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 256,
+ "final_success": 1.0,
+ "late_success": 1.0,
+ "learning_gain": 1.0,
+ "role_cosine_after_training": 1.0000001192092896,
+ "training_wall_s": 0.18617762252688408
+ },
+ "outcome_training_lesion": {
+ "cost": {
+ "active_state_episode_steps": 18917,
+ "cursor_scalar_observations": 9704,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 4852,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.17078045010566711,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.03125,
+ 0.03125,
+ 0.078125,
+ 0.265625,
+ 0.5,
+ 0.78125,
+ 0.84375,
+ 0.953125,
+ 0.9375,
+ 0.984375
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 1684,
+ "final_success": 0.96875,
+ "late_success": 0.9583333333333334,
+ "learning_gain": 0.9583333333333334,
+ "role_cosine_after_training": 0.9445701241493225,
+ "training_wall_s": 0.21972168236970901
+ },
+ "plasticity_lesion": {
+ "cost": {
+ "active_state_episode_steps": 25088,
+ "cursor_scalar_observations": 12544,
+ "maximum_state_episode_steps": 25088,
+ "reverse_mode_calls": 0,
+ "role_probe_examples": 6272,
+ "task_loss_queries": 0,
+ "terminal_outcome_observations": 896
+ },
+ "critic_l2_after_training": 0.06252607703208923,
+ "daily_success": [
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0,
+ 0.0
+ ],
+ "early_success": 0.0,
+ "evaluation_active_state_episode_steps": 7168,
+ "final_success": 0.0,
+ "late_success": 0.0,
+ "learning_gain": 0.0,
+ "role_cosine_after_training": 0.9957386255264282,
+ "training_wall_s": 0.1857893168926239
+ }
+ },
+ "config": {
+ "context_ar": 0.8,
+ "context_dim": 16,
+ "coupling_scale": 1.0,
+ "critic_eta": 0.03,
+ "days": 14,
+ "eligibility_decay": 0.8,
+ "episodes_per_day": 64,
+ "feedback": "performance_velocity",
+ "forward_eta": 0.1,
+ "gamma": 0.8,
+ "inertia": 0.65,
+ "kappa": 0.0,
+ "n_background": 30,
+ "n_minus": 5,
+ "n_plus": 5,
+ "perturb_every": 4,
+ "perturb_sigma": 0.03,
+ "predictor_eta": 0.2,
+ "process_noise": 0.12,
+ "steps_per_episode": 28,
+ "target": 0.8,
+ "terminal_reward": 1.0,
+ "vectorizer_eta": 0.03,
+ "velocity_reward_scale": 1.0
+ },
+ "finite": true,
+ "hardware": {
+ "device": "cpu",
+ "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35",
+ "threads": 1,
+ "torch_version": "2.10.0+cu128"
+ },
+ "peak_rss_mib": 954.734375,
+ "protocol": {
+ "calibration_uses_outcome_labels": false,
+ "confirmation_seeds_touched": false,
+ "failed_target_gate_sha256": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1",
+ "fixed_config": {
+ "critic_eta": 0.03,
+ "forward_eta": 0.1,
+ "gamma": 0.8,
+ "velocity_reward_scale": 1.0
+ },
+ "hyperparameter_selection": false,
+ "name": "oral_b_v2_calibrated_recovery_development_v1",
+ "protocol_sha256": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8",
+ "selection_split": "development_validation"
+ },
+ "provenance": {
+ "git_commit": "d289a46293e0b422c195746ad8b310d4b50ded23",
+ "git_tracked_dirty": false,
+ "input_sha256": {
+ "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc",
+ "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5",
+ "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c",
+ "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1",
+ "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef",
+ "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4",
+ "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1",
+ "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd",
+ "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597",
+ "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8",
+ "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b",
+ "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2",
+ "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa",
+ "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a",
+ "runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18",
+ "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae",
+ "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee"
+ },
+ "tracked_inputs": {
+ "base_dynamics": true,
+ "common_runner": true,
+ "confirmation_analyzer": true,
+ "confirmation_runner": true,
+ "d4_gate": true,
+ "development_analyzer": true,
+ "failed_target_gate": true,
+ "failed_v2_gate": true,
+ "old_r2_gate": true,
+ "protocol": true,
+ "recovery_confirmation_analyzer": true,
+ "recovery_development_analyzer": true,
+ "recovery_metrics": true,
+ "recovery_runner": true,
+ "runner": true,
+ "v2_dynamics": true,
+ "v2_metrics": true
+ }
+ },
+ "schema_version": 4,
+ "signatures": {
+ "acute_outcome_lesion_outcome_balanced_acc": 0.540903540903541,
+ "acute_outcome_lesion_role_aligned_separation": 0.010854148777368705,
+ "causal_role_sign_inversion_index": 0.03921735111033796,
+ "challenge_episodes": 640,
+ "challenge_failure_count": 315,
+ "challenge_success_count": 325,
+ "challenge_success_fraction": 0.5078125,
+ "critic_contribution_value_prediction_corr": 0.9999999999994963,
+ "decoder_distance_residual_corr": 0.13642571216406077,
+ "mean_abs_raw_soma_corr": 0.9990992136226702,
+ "mean_abs_residual_soma_corr": 0.05195367815050812,
+ "mean_critic_expectedness_contribution": 0.24424682945902534,
+ "nonterminal_training_events": 13328,
+ "raw_minus_residual_abs_soma_corr": 0.9471455354721621,
+ "role_aligned_error_cv_corr": 0.34897921307347385,
+ "role_aligned_velocity_cv_corr": 0.9982339724621896,
+ "surrounding_event_decoder_balanced_acc": 0.5528030824638526,
+ "target_success_fraction": {
+ "1.8872586727142333": 0.78125,
+ "1.9072148621082305": 0.640625,
+ "1.9221965074539185": 0.4765625,
+ "1.9328129887580872": 0.40625,
+ "1.9453917741775513": 0.234375
+ },
+ "terminal_outcome_separation_drop_under_acute_lesion": 0.38343148953250883,
+ "terminal_previous_soma_outcome_balanced_acc": 0.7603663003663004,
+ "terminal_residual_minus_previous_soma_acc": 0.23804639804639804,
+ "terminal_residual_outcome_balanced_acc": 0.9984126984126984,
+ "terminal_role_aligned_outcome_separation": 0.39428563830987756,
+ "terminal_training_events": 896,
+ "velocity_minus_error_abs_cv_corr": 0.6492547593887157
+ },
+ "split": "development_validation",
+ "wall_s": 2.422760460525751,
+ "warmup": {
+ "critic_training_lesion": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602800,
+ "role_cosine_after_warmup": 0.9957716464996338,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": true
+ },
+ "fixed_role": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602800,
+ "role_cosine_after_warmup": -0.1393236517906189,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": false
+ },
+ "intact": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602800,
+ "role_cosine_after_warmup": 0.9957716464996338,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": true
+ },
+ "oracle_role": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602800,
+ "role_cosine_after_warmup": 1.0000001192092896,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": false
+ },
+ "outcome_training_lesion": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602800,
+ "role_cosine_after_warmup": 0.9957716464996338,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": true
+ },
+ "plasticity_lesion": {
+ "batches": 100,
+ "examples": 6400,
+ "instruction_present": false,
+ "predictor_max_abs_error": 2.384185791015625e-07,
+ "rng_seed": 602800,
+ "role_cosine_after_warmup": 0.9957716464996338,
+ "role_cursor_scalar_observations": 12800,
+ "role_update_enabled": true
+ }
+ }
+}
diff --git a/results/bci_v2_calibrated_dev_gate.json b/results/bci_v2_calibrated_dev_gate.json
new file mode 100644
index 0000000..515f0e6
--- /dev/null
+++ b/results/bci_v2_calibrated_dev_gate.json
@@ -0,0 +1,148 @@
+{
+ "calibrated_confirmation_opened": true,
+ "calibrated_targets_by_task_seed": {
+ "26": [
+ 1.859693741798401,
+ 1.8864485919475555,
+ 1.9004749059677124,
+ 1.9156511783599854,
+ 1.929451322555542
+ ],
+ "27": [
+ 1.7139501571655273,
+ 1.7322039246559142,
+ 1.750077724456787,
+ 1.7649245738983155,
+ 1.783210277557373
+ ],
+ "28": [
+ 1.8872586727142333,
+ 1.9072148621082305,
+ 1.9221965074539185,
+ 1.9328129887580872,
+ 1.9453917741775513
+ ]
+ },
+ "calibration_quantiles": [
+ 0.2,
+ 0.35,
+ 0.5,
+ 0.65,
+ 0.8
+ ],
+ "challenge_success_fraction_by_task_seed": [
+ 0.5078125,
+ 0.4875,
+ 0.5078125
+ ],
+ "checks_by_task_seed": {
+ "26": {
+ "challenge_fraction_between_0p15_and_0p85": true,
+ "critic_contribution_value_corr_at_least_0p95": true,
+ "critic_expectedness_at_least_0p05": true,
+ "decoder_distance_corr_at_least_0p02": true,
+ "final_success_at_least_0p70": true,
+ "fixed_role_gap_at_least_0p20": true,
+ "learning_gain_at_least_0p10": true,
+ "oracle_deficit_at_most_0p10": true,
+ "outcome_lesion_separation_drop_at_least_0p20": true,
+ "plasticity_lesion_retains_at_most_half_gain": true,
+ "raw_residual_corr_gap_at_least_0p20": true,
+ "residual_soma_corr_at_most_0p10": true,
+ "role_cosine_at_least_0p80": true,
+ "sign_inversion_at_least_0p01": true,
+ "surrounding_event_accuracy_at_least_0p52": true,
+ "terminal_outcome_accuracy_at_least_0p75": true,
+ "terminal_role_separation_at_least_0p20": true,
+ "velocity_advantage_at_least_0p05": true
+ },
+ "27": {
+ "challenge_fraction_between_0p15_and_0p85": true,
+ "critic_contribution_value_corr_at_least_0p95": true,
+ "critic_expectedness_at_least_0p05": true,
+ "decoder_distance_corr_at_least_0p02": true,
+ "final_success_at_least_0p70": true,
+ "fixed_role_gap_at_least_0p20": true,
+ "learning_gain_at_least_0p10": true,
+ "oracle_deficit_at_most_0p10": true,
+ "outcome_lesion_separation_drop_at_least_0p20": true,
+ "plasticity_lesion_retains_at_most_half_gain": true,
+ "raw_residual_corr_gap_at_least_0p20": true,
+ "residual_soma_corr_at_most_0p10": true,
+ "role_cosine_at_least_0p80": true,
+ "sign_inversion_at_least_0p01": true,
+ "surrounding_event_accuracy_at_least_0p52": true,
+ "terminal_outcome_accuracy_at_least_0p75": true,
+ "terminal_role_separation_at_least_0p20": true,
+ "velocity_advantage_at_least_0p05": true
+ },
+ "28": {
+ "challenge_fraction_between_0p15_and_0p85": true,
+ "critic_contribution_value_corr_at_least_0p95": true,
+ "critic_expectedness_at_least_0p05": true,
+ "decoder_distance_corr_at_least_0p02": true,
+ "final_success_at_least_0p70": true,
+ "fixed_role_gap_at_least_0p20": true,
+ "learning_gain_at_least_0p10": true,
+ "oracle_deficit_at_most_0p10": true,
+ "outcome_lesion_separation_drop_at_least_0p20": true,
+ "plasticity_lesion_retains_at_most_half_gain": true,
+ "raw_residual_corr_gap_at_least_0p20": true,
+ "residual_soma_corr_at_most_0p10": true,
+ "role_cosine_at_least_0p80": true,
+ "sign_inversion_at_least_0p01": true,
+ "surrounding_event_accuracy_at_least_0p52": true,
+ "terminal_outcome_accuracy_at_least_0p75": true,
+ "terminal_role_separation_at_least_0p20": true,
+ "velocity_advantage_at_least_0p05": true
+ }
+ },
+ "complete_grid": true,
+ "confirmation_seeds_touched": false,
+ "fixed_config": {
+ "critic_eta": 0.03,
+ "forward_eta": 0.1,
+ "gamma": 0.8,
+ "velocity_reward_scale": 1.0
+ },
+ "input_sha256": {
+ "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc",
+ "common_runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5",
+ "confirmation_analyzer": "2536fa01f62c3b2d87e782b08075cb5948aa26bfdce71a1d8c2446d5b68bb34c",
+ "confirmation_runner": "5e9beba106b4b29c4ee1568052d6e8c15690dbde92917a2497a52e43e7f557d1",
+ "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef",
+ "development_analyzer": "dde4a2e4d62f5f9da94357b388139c50251b6863f9c991f1a006e71eeb7953e4",
+ "failed_target_gate": "1df37c0b639542bbee3017da54306b0d572cbf96230428f142fafed139ac92f1",
+ "failed_v2_gate": "e53f46cf456d60ce4d33586ace4f4619fb1a31a58082b2b72943d4ed10cd25fd",
+ "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597",
+ "protocol": "a175dc80fdb7218c010953d5d7e3c77fee0690d5f10fabaf9d700bb8505acde8",
+ "recovery_confirmation_analyzer": "29a168143266bc7078e5229183bb7ea45283973592fecd1fcd715488f5d6242b",
+ "recovery_development_analyzer": "2768b877132ece6d612d2a5f1e483c495b277e26fdf54fb8a716aa8de6fd67e2",
+ "recovery_metrics": "371cb6a7e2192d25ddbfad69c0ce7a1150aa1c21ca98e141bedb2afc2acbdbaa",
+ "recovery_runner": "cea0ef658c698c376b1fda0cf2f7f910ee44f7bcde153f2fbaf18e66a2d3661a",
+ "runner": "446ad26fc015eb79d28e0bb442c8aef71bdd1d0320910494cac20f92e3c7dc18",
+ "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae",
+ "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee"
+ },
+ "intact_final_by_task_seed": [
+ 1.0,
+ 1.0,
+ 1.0
+ ],
+ "protocol": "oral_b_v2_calibrated_recovery_development_v1",
+ "review_score_after": 7,
+ "review_score_before": 7,
+ "score_change_rule": "development validation never changes the formal score",
+ "source_commit": "d289a46293e0b422c195746ad8b310d4b50ded23",
+ "source_sha256": {
+ "results/bci_v2_calibrated_dev/bci_v2_calibrated_t26_m0.json": "3ad58c0d4ee6aef1835f281c60c63b7c309ff23971692ba97631fd81d43c7165",
+ "results/bci_v2_calibrated_dev/bci_v2_calibrated_t27_m0.json": "3c61254dbf22ba1cb550f4c35aecba2375c754ac64ad383c642ecaaff8085670",
+ "results/bci_v2_calibrated_dev/bci_v2_calibrated_t28_m0.json": "b09933f8776ae1941adb0db543d1f5df91e9287cb11b2536e27e3b9caafc5812"
+ },
+ "status": "passed",
+ "terminal_outcome_accuracy_by_task_seed": [
+ 0.9952380952380953,
+ 1.0,
+ 0.9984126984126984
+ ]
+}