diff options
| author | YurenHao0426 <Blackhao0426@gmail.com> | 2026-07-23 08:00:45 -0500 |
|---|---|---|
| committer | YurenHao0426 <Blackhao0426@gmail.com> | 2026-07-23 08:00:45 -0500 |
| commit | eb021a6beac9d4b55a595171f149d669868c403b (patch) | |
| tree | f66f2b75d560f8305ac49c8301cb5cddda87d0a2 | |
| parent | 670c1399659f31ac79c4b454147417d4f8ddca7d (diff) | |
results: retain failed oral-B-v2 development grid
25 files changed, 10877 insertions, 0 deletions
diff --git a/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p01_t20_m0.json b/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p01_t20_m0.json new file mode 100644 index 0000000..4cae13c --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p01_t20_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.01, + "forward_eta": 0.03, + "gamma": 0.8, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 20 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 430002, + "16": 430003, + "20": 430004, + "24": 430005, + "28": 430006, + "4": 430000, + "8": 430001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400020 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949056506156921, + "training_wall_s": 0.1929846741259098 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006131102796643972, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.20932374894618988 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006262489128857851, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949058294296265, + "training_wall_s": 0.2878704331815243 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.00626755366101861, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.2164529636502266 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006262489128857851, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949058294296265, + "training_wall_s": 0.210897047072649 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006139965727925301, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.18343744054436684 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.01, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.03, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 944.46484375, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0004354078803623657, + "causal_role_sign_inversion_index": 0.007238296215345278, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999713596, + "decoder_distance_residual_corr": 0.29853281220448974, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999687757945163, + "mean_abs_residual_soma_corr": 0.03269399053748619, + "mean_critic_expectedness_contribution": -0.000423994680073208, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.96727478525703, + "role_aligned_error_cv_corr": 0.3096127276417904, + "role_aligned_velocity_cv_corr": 0.999763879701073, + "surrounding_event_decoder_balanced_acc": 0.596699684698489, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0004354078803623657, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6901511520592827 + }, + "split": "development", + "wall_s": 2.7543775103986263, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p01_t21_m0.json b/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p01_t21_m0.json new file mode 100644 index 0000000..756313c --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p01_t21_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.01, + "forward_eta": 0.03, + "gamma": 0.8, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 21 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 431002, + "16": 431003, + "20": 431004, + "24": 431005, + "28": 431006, + "4": 431000, + "8": 431001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400021 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907992362976, + "training_wall_s": 0.2123476304113865 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.005946753546595573, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21735524758696556 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006090307142585516, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.2924068532884121 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006094345822930336, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.23121720552444458 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006090307142585516, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.23432151228189468 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.005957168992608786, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.20393254235386848 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.01, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.03, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 956.22265625, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.00041215065973168163, + "causal_role_sign_inversion_index": 0.007302106876618754, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999678241, + "decoder_distance_residual_corr": 0.3007753755400901, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999695364758214, + "mean_abs_residual_soma_corr": 0.03433453004426391, + "mean_critic_expectedness_contribution": -0.0004537493910206791, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9656350064315575, + "role_aligned_error_cv_corr": 0.30782991000236093, + "role_aligned_velocity_cv_corr": 0.9997131247883869, + "surrounding_event_decoder_balanced_acc": 0.5982113364506751, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.00041215065973168163, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.691883214786026 + }, + "split": "development", + "wall_s": 2.8594805151224136, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p01_t22_m0.json b/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p01_t22_m0.json new file mode 100644 index 0000000..11f7a61 --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p01_t22_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.01, + "forward_eta": 0.03, + "gamma": 0.8, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 22 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 432002, + "16": 432003, + "20": 432004, + "24": 432005, + "28": 432006, + "4": 432000, + "8": 432001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400022 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.1943834200501442 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.005971681792289019, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.2104893997311592 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006062759086489677, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946003556251526, + "training_wall_s": 0.2892235219478607 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006065902300179005, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.2102864608168602 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006062759086489677, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946003556251526, + "training_wall_s": 0.2120020017027855 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.005976130720227957, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.1847543604671955 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.01, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.03, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 951.10546875, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0005991807837899413, + "causal_role_sign_inversion_index": 0.007111037277678306, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999679303, + "decoder_distance_residual_corr": 0.29922582814085646, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999702832306415, + "mean_abs_residual_soma_corr": 0.03457850645999337, + "mean_critic_expectedness_contribution": -0.0004458842080274433, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9653917767706481, + "role_aligned_error_cv_corr": 0.30824265131321216, + "role_aligned_velocity_cv_corr": 0.9998059457297688, + "surrounding_event_decoder_balanced_acc": 0.5948567288210119, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0005991807837899413, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6915632944165566 + }, + "split": "development", + "wall_s": 2.8021088913083076, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p03_t20_m0.json b/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p03_t20_m0.json new file mode 100644 index 0000000..d292676 --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p03_t20_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.03, + "forward_eta": 0.03, + "gamma": 0.8, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 20 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 430002, + "16": 430003, + "20": 430004, + "24": 430005, + "28": 430006, + "4": 430000, + "8": 430001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400020 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949056506156921, + "training_wall_s": 0.19733485206961632 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015812627971172333, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21377084404230118 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.016062837094068527, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949056506156921, + "training_wall_s": 0.2931537888944149 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.016071317717432976, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.21492749452590942 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.016062837094068527, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949056506156921, + "training_wall_s": 0.21800879389047623 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015832092612981796, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.18806280568242073 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.03, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 957.9765625, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0005316698106072761, + "causal_role_sign_inversion_index": 0.0070416286238234385, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999954686, + "decoder_distance_residual_corr": 0.2872360328255551, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999705222405215, + "mean_abs_residual_soma_corr": 0.03519705816039624, + "mean_critic_expectedness_contribution": -0.0005261400100045056, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9647734640801253, + "role_aligned_error_cv_corr": 0.3243090751578382, + "role_aligned_velocity_cv_corr": 0.9991768227284149, + "surrounding_event_decoder_balanced_acc": 0.5917881015778483, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0005316698106072761, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6748677475705767 + }, + "split": "development", + "wall_s": 2.8073874935507774, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p03_t21_m0.json b/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p03_t21_m0.json new file mode 100644 index 0000000..28ee17b --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p03_t21_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.03, + "forward_eta": 0.03, + "gamma": 0.8, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 21 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 431002, + "16": 431003, + "20": 431004, + "24": 431005, + "28": 431006, + "4": 431000, + "8": 431001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400021 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907992362976, + "training_wall_s": 0.21300701051950455 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015472296625375748, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.23187587037682533 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015740016475319862, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907992362976, + "training_wall_s": 0.3138655684888363 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01574784703552723, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.23042933642864227 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015740016475319862, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907992362976, + "training_wall_s": 0.2343682087957859 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015492268837988377, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.2031250260770321 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.03, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 958.03125, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.00047288746003096304, + "causal_role_sign_inversion_index": 0.007109758000545506, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999948241, + "decoder_distance_residual_corr": 0.2895889569047025, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999711990262169, + "mean_abs_residual_soma_corr": 0.036869044032326506, + "mean_critic_expectedness_contribution": -0.0005059797267910526, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9631021549938904, + "role_aligned_error_cv_corr": 0.32164838023935644, + "role_aligned_velocity_cv_corr": 0.9991556608551725, + "surrounding_event_decoder_balanced_acc": 0.5942504896430193, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.00047288746003096304, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.677507280615816 + }, + "split": "development", + "wall_s": 2.904862642288208, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p03_t22_m0.json b/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p03_t22_m0.json new file mode 100644 index 0000000..d52de4e --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p03_t22_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.03, + "forward_eta": 0.03, + "gamma": 0.8, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 22 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 432002, + "16": 432003, + "20": 432004, + "24": 432005, + "28": 432006, + "4": 432000, + "8": 432001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400022 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.19646292552351952 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015561887063086033, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21270771697163582 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015729214996099472, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946001768112183, + "training_wall_s": 0.2937253527343273 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015735017135739326, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.21198097616434097 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015729214996099472, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946001768112183, + "training_wall_s": 0.2144930697977543 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015574400313198566, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.18628036975860596 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.03, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 956.6328125, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0006540793102450603, + "causal_role_sign_inversion_index": 0.006921697694125522, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999946487, + "decoder_distance_residual_corr": 0.2884725390273238, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999719075414306, + "mean_abs_residual_soma_corr": 0.036882013566283386, + "mean_critic_expectedness_contribution": -0.0004914866783506923, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9630898939751472, + "role_aligned_error_cv_corr": 0.321477908768127, + "role_aligned_velocity_cv_corr": 0.999024694980346, + "surrounding_event_decoder_balanced_acc": 0.5906992162345662, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0006540793102450603, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.677546786212219 + }, + "split": "development", + "wall_s": 2.814539149403572, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p01_t20_m0.json b/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p01_t20_m0.json new file mode 100644 index 0000000..6c38d27 --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p01_t20_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.01, + "forward_eta": 0.03, + "gamma": 0.95, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 20 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 430002, + "16": 430003, + "20": 430004, + "24": 430005, + "28": 430006, + "4": 430000, + "8": 430001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400020 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949056506156921, + "training_wall_s": 0.19653043150901794 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006414641626179218, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21637140214443207 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006557112094014883, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.29182596504688263 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0065628099255263805, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.21889567375183105 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006557112094014883, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.22790521383285522 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.00642406614497304, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.1899530440568924 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.01, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.03, + "gamma": 0.95, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 958.296875, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0006096375123782318, + "causal_role_sign_inversion_index": 0.007202363651987183, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999727303, + "decoder_distance_residual_corr": 0.30429553352922695, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.999969040995053, + "mean_abs_residual_soma_corr": 0.031503231471301955, + "mean_critic_expectedness_contribution": -0.0005980293817092118, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.968465809523751, + "role_aligned_error_cv_corr": 0.3029221079064835, + "role_aligned_velocity_cv_corr": 0.9998278584783713, + "surrounding_event_decoder_balanced_acc": 0.5993587601218684, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0006096375123782318, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6969057505718879 + }, + "split": "development", + "wall_s": 2.867789041250944, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p01_t21_m0.json b/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p01_t21_m0.json new file mode 100644 index 0000000..4b60dbb --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p01_t21_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.01, + "forward_eta": 0.03, + "gamma": 0.95, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 21 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 431002, + "16": 431003, + "20": 431004, + "24": 431005, + "28": 431006, + "4": 431000, + "8": 431001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400021 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907992362976, + "training_wall_s": 0.19531415030360222 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006219979841262102, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21133632957935333 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006383350118994713, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907992362976, + "training_wall_s": 0.29000579565763474 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0063881403766572475, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.210700161755085 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006383350118994713, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907992362976, + "training_wall_s": 0.21354755386710167 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006231742445379496, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.1847149208188057 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.01, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.03, + "gamma": 0.95, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 946.07421875, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.000604763396529065, + "causal_role_sign_inversion_index": 0.007269241648451328, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.99999999996943, + "decoder_distance_residual_corr": 0.30617528169333064, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999697672957018, + "mean_abs_residual_soma_corr": 0.03305602032832172, + "mean_critic_expectedness_contribution": -0.0006513508007891453, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9669137469673801, + "role_aligned_error_cv_corr": 0.30161111933961604, + "role_aligned_velocity_cv_corr": 0.9997637893511634, + "surrounding_event_decoder_balanced_acc": 0.5994279484525389, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.000604763396529065, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6981526700115473 + }, + "split": "development", + "wall_s": 2.7387674786150455, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p01_t22_m0.json b/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p01_t22_m0.json new file mode 100644 index 0000000..7d4ede6 --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p01_t22_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.01, + "forward_eta": 0.03, + "gamma": 0.95, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 22 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 432002, + "16": 432003, + "20": 432004, + "24": 432005, + "28": 432006, + "4": 432000, + "8": 432001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400022 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.19748739153146744 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006243748124688864, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21457795798778534 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006349558010697365, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946003556251526, + "training_wall_s": 0.29552317410707474 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006353338249027729, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.21326586604118347 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006349558010697365, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946003556251526, + "training_wall_s": 0.21588260680437088 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006248936522752047, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.18914302438497543 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.01, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.03, + "gamma": 0.95, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 958.05078125, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0007898118119727295, + "causal_role_sign_inversion_index": 0.0070816839452064375, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999664465, + "decoder_distance_residual_corr": 0.3048428552841237, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.999970490594729, + "mean_abs_residual_soma_corr": 0.033486957811605954, + "mean_critic_expectedness_contribution": -0.0006430637122453183, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.966483532783123, + "role_aligned_error_cv_corr": 0.3021537555385858, + "role_aligned_velocity_cv_corr": 0.9998357741243664, + "surrounding_event_decoder_balanced_acc": 0.5965125613997466, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0007898118119727295, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6976820185857806 + }, + "split": "development", + "wall_s": 2.8262000381946564, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p03_t20_m0.json b/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p03_t20_m0.json new file mode 100644 index 0000000..84e8311 --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p03_t20_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.03, + "forward_eta": 0.03, + "gamma": 0.95, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 20 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 430002, + "16": 430003, + "20": 430004, + "24": 430005, + "28": 430006, + "4": 430000, + "8": 430001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400020 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949056506156921, + "training_wall_s": 0.20568963512778282 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.017393946647644043, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21949751675128937 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.017733843997120857, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.2983199656009674 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0177466981112957, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.2240881212055683 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.017733843997120857, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.23205474764108658 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.017417721450328827, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.19746868312358856 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.03, + "gamma": 0.95, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 956.92578125, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0011786516144934069, + "causal_role_sign_inversion_index": 0.006927962132847491, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999959622, + "decoder_distance_residual_corr": 0.3034607981139468, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999713577627342, + "mean_abs_residual_soma_corr": 0.03201802183628514, + "mean_critic_expectedness_contribution": -0.0011675952718466452, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.967953335926449, + "role_aligned_error_cv_corr": 0.3064079878090659, + "role_aligned_velocity_cv_corr": 0.9993464738605745, + "surrounding_event_decoder_balanced_acc": 0.5986790965304201, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0011786516144934069, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6929384860515087 + }, + "split": "development", + "wall_s": 2.8720116317272186, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p03_t21_m0.json b/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p03_t21_m0.json new file mode 100644 index 0000000..60a1542 --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p03_t21_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.03, + "forward_eta": 0.03, + "gamma": 0.95, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 21 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 431002, + "16": 431003, + "20": 431004, + "24": 431005, + "28": 431006, + "4": 431000, + "8": 431001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400021 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907992362976, + "training_wall_s": 0.1947210542857647 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.016924316063523293, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21093566343188286 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01730191335082054, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.29469459876418114 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01731295883655548, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.22063887864351273 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01730191335082054, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.22155699878931046 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01695229485630989, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.18521039187908173 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.03, + "gamma": 0.95, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 957.2578125, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0012100607993267075, + "causal_role_sign_inversion_index": 0.007005868556329141, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999959489, + "decoder_distance_residual_corr": 0.3050653599857341, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999719343333229, + "mean_abs_residual_soma_corr": 0.03349548624573582, + "mean_critic_expectedness_contribution": -0.001250195367026001, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.966476448087587, + "role_aligned_error_cv_corr": 0.3051054801685798, + "role_aligned_velocity_cv_corr": 0.9992939819771297, + "surrounding_event_decoder_balanced_acc": 0.5995547470516674, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0012100607993267075, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6941885018085499 + }, + "split": "development", + "wall_s": 2.8191695734858513, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p03_t22_m0.json b/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p03_t22_m0.json new file mode 100644 index 0000000..7dc5547 --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p03_t22_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.03, + "forward_eta": 0.03, + "gamma": 0.95, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 22 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 432002, + "16": 432003, + "20": 432004, + "24": 432005, + "28": 432006, + "4": 432000, + "8": 432001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400022 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.21405668184161186 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01700076460838318, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21256864443421364 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.017237728461623192, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946003556251526, + "training_wall_s": 0.2932116650044918 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01724599301815033, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.21187813952565193 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.017237728461623192, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946003556251526, + "training_wall_s": 0.2132866345345974 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01701347716152668, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.18623150512576103 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.03, + "gamma": 0.95, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 946.1875, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0013960065261283611, + "causal_role_sign_inversion_index": 0.006826995555184645, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999957648, + "decoder_distance_residual_corr": 0.3034612647605678, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999725759462719, + "mean_abs_residual_soma_corr": 0.03396802618204717, + "mean_critic_expectedness_contribution": -0.0012419408459255475, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9660045497642247, + "role_aligned_error_cv_corr": 0.305279631121636, + "role_aligned_velocity_cv_corr": 0.9990713932403695, + "surrounding_event_decoder_balanced_acc": 0.596085308196203, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0013960065261283611, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6937917621187335 + }, + "split": "development", + "wall_s": 2.79043235629797, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p01_t20_m0.json b/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p01_t20_m0.json new file mode 100644 index 0000000..652299d --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p01_t20_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.01, + "forward_eta": 0.1, + "gamma": 0.8, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 20 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 430002, + "16": 430003, + "20": 430004, + "24": 430005, + "28": 430006, + "4": 430000, + "8": 430001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400020 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.21380134299397469 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006111179944127798, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.23342447727918625 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006634820718318224, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949056506156921, + "training_wall_s": 0.31351136788725853 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006658447906374931, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.23457582294940948 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006634820718318224, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949056506156921, + "training_wall_s": 0.23702435567975044 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006139965727925301, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.20495665073394775 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.01, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 952.03125, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.00044206465441409175, + "causal_role_sign_inversion_index": 0.007549123420835249, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999790935, + "decoder_distance_residual_corr": 0.2770667242352346, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.999966712752928, + "mean_abs_residual_soma_corr": 0.035164190106566254, + "mean_critic_expectedness_contribution": -0.00044276197136624624, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9648025226463618, + "role_aligned_error_cv_corr": 0.29908938396430934, + "role_aligned_velocity_cv_corr": 0.9997547027286591, + "surrounding_event_decoder_balanced_acc": 0.5884033700057648, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.00044206465441409175, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.7006653187643497 + }, + "split": "development", + "wall_s": 3.032918691635132, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p01_t21_m0.json b/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p01_t21_m0.json new file mode 100644 index 0000000..313eb41 --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p01_t21_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.01, + "forward_eta": 0.1, + "gamma": 0.8, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 21 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 431002, + "16": 431003, + "20": 431004, + "24": 431005, + "28": 431006, + "4": 431000, + "8": 431001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400021 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951906800270081, + "training_wall_s": 0.19734221696853638 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.005923416465520859, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21407122910022736 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006500380579382181, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.29416946321725845 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006519318092614412, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.2124783918261528 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006500380579382181, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.21469618752598763 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.005957168992608786, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.18629001453518867 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.01, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 957.28515625, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.00045068150644091864, + "causal_role_sign_inversion_index": 0.007636555881307667, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999772188, + "decoder_distance_residual_corr": 0.2796288062249614, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.99996734386997, + "mean_abs_residual_soma_corr": 0.03681060901953441, + "mean_critic_expectedness_contribution": -0.00048221625182699897, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9631567348504356, + "role_aligned_error_cv_corr": 0.2959652964685003, + "role_aligned_velocity_cv_corr": 0.9997041147219436, + "surrounding_event_decoder_balanced_acc": 0.5905475893466077, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.00045068150644091864, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.7037388182534434 + }, + "split": "development", + "wall_s": 2.821523994207382, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p01_t22_m0.json b/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p01_t22_m0.json new file mode 100644 index 0000000..e53cdea --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p01_t22_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.01, + "forward_eta": 0.1, + "gamma": 0.8, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 22 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 432002, + "16": 432003, + "20": 432004, + "24": 432005, + "28": 432006, + "4": 432000, + "8": 432001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400022 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946003556251526, + "training_wall_s": 0.19723769649863243 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.005961690563708544, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21356197074055672 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006327390670776367, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.2945389784872532 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006342768669128418, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.21401751786470413 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006327390670776367, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.21582948043942451 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.005976130720227957, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.1882387474179268 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.01, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 957.43359375, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0006100508636421717, + "causal_role_sign_inversion_index": 0.007347918307751283, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999740408, + "decoder_distance_residual_corr": 0.2828764761714883, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999687281462366, + "mean_abs_residual_soma_corr": 0.036753978975501776, + "mean_critic_expectedness_contribution": -0.0004547985563814416, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9632147491707348, + "role_aligned_error_cv_corr": 0.29900114530991945, + "role_aligned_velocity_cv_corr": 0.999788807067824, + "surrounding_event_decoder_balanced_acc": 0.5904555836274331, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0006100508636421717, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.7007876617579045 + }, + "split": "development", + "wall_s": 2.763981442898512, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p03_t20_m0.json b/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p03_t20_m0.json new file mode 100644 index 0000000..db0abef --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p03_t20_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 20 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14293, + "acute_outcome_lesion": 14293, + "intact": 14293 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 430002, + "16": 430003, + "20": 430004, + "24": 430005, + "28": 430006, + "4": 430000, + "8": 430001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400020 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.19959724694490433 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01576717011630535, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.20899632945656776 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25079, + "cursor_scalar_observations": 12540, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6270, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01584676466882229, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.015625, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7160, + "final_success": 0.00390625, + "late_success": 0.005208333333333333, + "learning_gain": 0.005208333333333333, + "role_cosine_after_training": 0.9948381185531616, + "training_wall_s": 0.36351219937205315 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25064, + "cursor_scalar_observations": 12534, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6267, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015464222989976406, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.03125, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7156, + "final_success": 0.01171875, + "late_success": 0.010416666666666666, + "learning_gain": 0.010416666666666666, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.21819698065519333 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25079, + "cursor_scalar_observations": 12540, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6270, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01674545742571354, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9948381185531616, + "training_wall_s": 0.23069345578551292 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015832092612981796, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.18239442631602287 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 956.125, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.7006734006734007, + "acute_outcome_lesion_role_aligned_separation": 0.011059083348233974, + "causal_role_sign_inversion_index": 0.007590452460995779, + "challenge_episodes": 896, + "challenge_failure_count": 891, + "challenge_success_count": 5, + "challenge_success_fraction": 0.005580357142857143, + "critic_contribution_value_prediction_corr": 0.9999999999968779, + "decoder_distance_residual_corr": 0.2411186706053349, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0078125, + "20": 0.0078125, + "24": 0.0078125, + "28": 0.015625, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999668350668788, + "mean_abs_residual_soma_corr": 0.04008589505926817, + "mean_critic_expectedness_contribution": -0.0001966855914630647, + "nonterminal_training_events": 24183, + "raw_minus_residual_abs_soma_corr": 0.9598809400076106, + "role_aligned_error_cv_corr": 0.3137738025773892, + "role_aligned_velocity_cv_corr": 0.999093314421921, + "surrounding_event_decoder_balanced_acc": 0.5778952070596446, + "terminal_outcome_separation_drop_under_acute_lesion": 0.4018947298920858, + "terminal_previous_soma_outcome_balanced_acc": 0.5837261503928171, + "terminal_residual_minus_previous_soma_acc": 0.41627384960718294, + "terminal_residual_outcome_balanced_acc": 1.0, + "terminal_role_aligned_outcome_separation": 0.41295381324031977, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6853195118445318 + }, + "split": "development", + "wall_s": 2.9079671017825603, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p03_t21_m0.json b/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p03_t21_m0.json new file mode 100644 index 0000000..3e91ce4 --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p03_t21_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 21 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14331, + "acute_outcome_lesion": 14331, + "intact": 14331 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 431002, + "16": 431003, + "20": 431004, + "24": 431005, + "28": 431006, + "4": 431000, + "8": 431001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400021 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951906800270081, + "training_wall_s": 0.19755160436034203 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015426043421030045, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.21122844889760017 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25083, + "cursor_scalar_observations": 12542, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6271, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015419727191329002, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0, + 0.015625 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7140, + "final_success": 0.0078125, + "late_success": 0.005208333333333333, + "learning_gain": 0.005208333333333333, + "role_cosine_after_training": 0.9951568245887756, + "training_wall_s": 0.29476047307252884 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25083, + "cursor_scalar_observations": 12542, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6271, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01545600313693285, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0, + 0.015625 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7140, + "final_success": 0.0078125, + "late_success": 0.005208333333333333, + "learning_gain": 0.005208333333333333, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.2154560163617134 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25083, + "cursor_scalar_observations": 12542, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6271, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01650770753622055, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.015625, + 0.0, + 0.0, + 0.015625 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7148, + "final_success": 0.00390625, + "late_success": 0.005208333333333333, + "learning_gain": 0.005208333333333333, + "role_cosine_after_training": 0.9951567053794861, + "training_wall_s": 0.21890860050916672 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015492268837988377, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.18639344349503517 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 957.15625, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.4664804469273743, + "acute_outcome_lesion_role_aligned_separation": 0.0004487390380411975, + "causal_role_sign_inversion_index": 0.0076279770769742735, + "challenge_episodes": 896, + "challenge_failure_count": 895, + "challenge_success_count": 1, + "challenge_success_fraction": 0.0011160714285714285, + "critic_contribution_value_prediction_corr": 0.9999999999962778, + "decoder_distance_residual_corr": 0.248516507556347, + "horizon_success_fraction": { + "12": 0.0078125, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.999967858030843, + "mean_abs_residual_soma_corr": 0.04170398694494832, + "mean_critic_expectedness_contribution": -0.00011620517373783239, + "nonterminal_training_events": 24187, + "raw_minus_residual_abs_soma_corr": 0.9582638710858946, + "role_aligned_error_cv_corr": 0.3097555622042727, + "role_aligned_velocity_cv_corr": 0.999053363357804, + "surrounding_event_decoder_balanced_acc": 0.5809516540632432, + "terminal_outcome_separation_drop_under_acute_lesion": 0.40037058890418087, + "terminal_previous_soma_outcome_balanced_acc": 0.4994413407821229, + "terminal_residual_minus_previous_soma_acc": 0.0005586592178771221, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": 0.4008193279422221, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6892978011535313 + }, + "split": "development", + "wall_s": 2.794892504811287, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p03_t22_m0.json b/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p03_t22_m0.json new file mode 100644 index 0000000..57f06f9 --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p03_t22_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.8, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 22 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 432002, + "16": 432003, + "20": 432004, + "24": 432005, + "28": 432006, + "4": 432000, + "8": 432001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400022 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946003556251526, + "training_wall_s": 0.21681366860866547 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015532732009887695, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.2341603972017765 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.016242388635873795, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.3125084228813648 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01627274975180626, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.23344024643301964 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.016242388635873795, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.23566758260130882 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.015574400313198566, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.2053239606320858 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.8, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 958.10546875, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0006112609351708791, + "causal_role_sign_inversion_index": 0.007273438124456074, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999966643, + "decoder_distance_residual_corr": 0.25635489692366603, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.999969695785536, + "mean_abs_residual_soma_corr": 0.04065599750786667, + "mean_critic_expectedness_contribution": -0.00041896542624211125, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9593136982776693, + "role_aligned_error_cv_corr": 0.3120558519374075, + "role_aligned_velocity_cv_corr": 0.9988698428175757, + "surrounding_event_decoder_balanced_acc": 0.5819735166629558, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0006112609351708791, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.6868139908801683 + }, + "split": "development", + "wall_s": 2.920739535242319, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p01_t20_m0.json b/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p01_t20_m0.json new file mode 100644 index 0000000..5b62e37 --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p01_t20_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.01, + "forward_eta": 0.1, + "gamma": 0.95, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 20 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 430002, + "16": 430003, + "20": 430004, + "24": 430005, + "28": 430006, + "4": 430000, + "8": 430001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400020 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.19619881361722946 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006393498741090298, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.211549062281847 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006954761687666178, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.29492445290088654 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006980822421610355, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.21154266223311424 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006954761687666178, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.21419654786586761 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.00642406614497304, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.1841624639928341 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.01, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.95, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 951.98828125, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0006487241034523411, + "causal_role_sign_inversion_index": 0.007451680412106713, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.999999999977282, + "decoder_distance_residual_corr": 0.28826336967261385, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999674274167039, + "mean_abs_residual_soma_corr": 0.03340999112965485, + "mean_critic_expectedness_contribution": -0.0006516016297714989, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.966557436287049, + "role_aligned_error_cv_corr": 0.29194125525372155, + "role_aligned_velocity_cv_corr": 0.9998316705758526, + "surrounding_event_decoder_balanced_acc": 0.5925177599355922, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0006487241034523411, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.707890415322131 + }, + "split": "development", + "wall_s": 2.7917996868491173, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p01_t21_m0.json b/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p01_t21_m0.json new file mode 100644 index 0000000..494b167 --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p01_t21_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.01, + "forward_eta": 0.1, + "gamma": 0.95, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 21 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 431002, + "16": 431003, + "20": 431004, + "24": 431005, + "28": 431006, + "4": 431000, + "8": 431001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400021 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951906800270081, + "training_wall_s": 0.2112559750676155 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0061936331912875175, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.2246597781777382 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006846026983112097, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951906800270081, + "training_wall_s": 0.2958148531615734 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0068686665035784245, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.22988076508045197 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006846026983112097, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951906800270081, + "training_wall_s": 0.23073595017194748 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006231742445379496, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.20061421021819115 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.01, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.95, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 951.1171875, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0006723258121977652, + "causal_role_sign_inversion_index": 0.007547032528277241, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999763568, + "decoder_distance_residual_corr": 0.2907599077715082, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999679506779687, + "mean_abs_residual_soma_corr": 0.03486272894487137, + "mean_critic_expectedness_contribution": -0.0007264005344150613, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9651052217330973, + "role_aligned_error_cv_corr": 0.28937076657613614, + "role_aligned_velocity_cv_corr": 0.9997657351027267, + "surrounding_event_decoder_balanced_acc": 0.594261657602474, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0006723258121977652, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.7103949685265906 + }, + "split": "development", + "wall_s": 2.9179469980299473, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p01_t22_m0.json b/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p01_t22_m0.json new file mode 100644 index 0000000..d29acff --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p01_t22_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.01, + "forward_eta": 0.1, + "gamma": 0.95, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 22 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 432002, + "16": 432003, + "20": 432004, + "24": 432005, + "28": 432006, + "4": 432000, + "8": 432001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400022 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946003556251526, + "training_wall_s": 0.19682318344712257 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.00623211357742548, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.2135714888572693 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006656359415501356, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.29799361154437065 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006674992386251688, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.21316815167665482 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006656359415501356, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.21628426387906075 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.006248936522752047, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.18778561055660248 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.01, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.95, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 958.52734375, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.000818418757894221, + "causal_role_sign_inversion_index": 0.0072731114768667384, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.999999999975314, + "decoder_distance_residual_corr": 0.29408184932024084, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999692332346296, + "mean_abs_residual_soma_corr": 0.03505511517300195, + "mean_critic_expectedness_contribution": -0.0006900817586476081, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9649141180616276, + "role_aligned_error_cv_corr": 0.29284382301771344, + "role_aligned_velocity_cv_corr": 0.9998277540416789, + "surrounding_event_decoder_balanced_acc": 0.593527945614895, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.000818418757894221, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.7069839310239654 + }, + "split": "development", + "wall_s": 2.8684824034571648, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p03_t20_m0.json b/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p03_t20_m0.json new file mode 100644 index 0000000..8d1dcad --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p03_t20_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.95, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 20 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 430002, + "16": 430003, + "20": 430004, + "24": 430005, + "28": 430006, + "4": 430000, + "8": 430001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400020 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.21185381338000298 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.017340010032057762, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.22840004041790962 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.018689222633838654, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949055910110474, + "training_wall_s": 0.2908512018620968 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01874866522848606, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.2290942706167698 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.018689222633838654, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949055910110474, + "training_wall_s": 0.23306182026863098 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.017417721450328827, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9949057102203369, + "training_wall_s": 0.19832388684153557 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.95, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 953.69921875, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0012213932154969294, + "causal_role_sign_inversion_index": 0.007207292572410673, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.999999999997275, + "decoder_distance_residual_corr": 0.2822848226229672, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999696526177205, + "mean_abs_residual_soma_corr": 0.03446035925618117, + "mean_critic_expectedness_contribution": -0.0012233987147360038, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9655092933615393, + "role_aligned_error_cv_corr": 0.2956435894253203, + "role_aligned_velocity_cv_corr": 0.9993065879475567, + "surrounding_event_decoder_balanced_acc": 0.5905286943117715, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0012213932154969294, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.7036629985222363 + }, + "split": "development", + "wall_s": 2.9077258370816708, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602000, + "role_cosine_after_warmup": 0.9944483041763306, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p03_t21_m0.json b/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p03_t21_m0.json new file mode 100644 index 0000000..0117fae --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p03_t21_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.95, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 21 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 431002, + "16": 431003, + "20": 431004, + "24": 431005, + "28": 431006, + "4": 431000, + "8": 431001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400021 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951906800270081, + "training_wall_s": 0.20855551213026047 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.016861194744706154, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.22409478947520256 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01837053708732128, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.29680583626031876 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.018421972170472145, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.22370364144444466 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01837053708732128, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.22836651280522346 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01695229485630989, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9951907396316528, + "training_wall_s": 0.19632023945450783 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.95, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 951.69140625, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.001303897821713616, + "causal_role_sign_inversion_index": 0.007306049294882835, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999972452, + "decoder_distance_residual_corr": 0.28449778845814067, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999701037037492, + "mean_abs_residual_soma_corr": 0.03588898135985759, + "mean_critic_expectedness_contribution": -0.001330264132719007, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9640811223438915, + "role_aligned_error_cv_corr": 0.2930031709964398, + "role_aligned_velocity_cv_corr": 0.9992079636917646, + "surrounding_event_decoder_balanced_acc": 0.592762168076139, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.001303897821713616, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.7062047926953248 + }, + "split": "development", + "wall_s": 2.830858774483204, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602100, + "role_cosine_after_warmup": 0.9963394403457642, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p03_t22_m0.json b/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p03_t22_m0.json new file mode 100644 index 0000000..ff643a6 --- /dev/null +++ b/results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p03_t22_m0.json @@ -0,0 +1,424 @@ +{ + "args": { + "critic_eta": 0.03, + "forward_eta": 0.1, + "gamma": 0.95, + "model_seed": 0, + "outdir": "results/bci_v2_dev", + "task_seed": 22 + }, + "assays": { + "challenge": { + "active_state_episode_steps_by_mode": { + "acute_critic_lesion": 14336, + "acute_outcome_lesion": 14336, + "intact": 14336 + }, + "episodes_per_horizon": 128, + "horizons": [ + 4, + 8, + 12, + 16, + 20, + 24, + 28 + ], + "selection_over_horizons": false, + "trajectory_seeds": { + "12": 432002, + "16": 432003, + "20": 432004, + "24": 432005, + "28": 432006, + "4": 432000, + "8": 432001 + } + }, + "performance_evaluation_episodes": 256, + "performance_evaluation_seed": 400022 + }, + "conditions": { + "critic_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.0, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946003556251526, + "training_wall_s": 0.20402583107352257 + }, + "fixed_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01697196066379547, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": -0.1393236517906189, + "training_wall_s": 0.22693512216210365 + }, + "intact": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.017918910831212997, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.30061565712094307 + }, + "oracle_role": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.017958827316761017, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 1.0000001192092896, + "training_wall_s": 0.2298455499112606 + }, + "outcome_training_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.017918910831212997, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.22780942171812057 + }, + "plasticity_lesion": { + "cost": { + "active_state_episode_steps": 25088, + "cursor_scalar_observations": 12544, + "maximum_state_episode_steps": 25088, + "reverse_mode_calls": 0, + "role_probe_examples": 6272, + "task_loss_queries": 0, + "terminal_outcome_observations": 896 + }, + "critic_l2_after_training": 0.01701347716152668, + "daily_success": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + "early_success": 0.0, + "evaluation_active_state_episode_steps": 7168, + "final_success": 0.0, + "late_success": 0.0, + "learning_gain": 0.0, + "role_cosine_after_training": 0.9946002960205078, + "training_wall_s": 0.19302869960665703 + } + }, + "config": { + "context_ar": 0.8, + "context_dim": 16, + "coupling_scale": 1.0, + "critic_eta": 0.03, + "days": 14, + "eligibility_decay": 0.8, + "episodes_per_day": 64, + "feedback": "performance_velocity", + "forward_eta": 0.1, + "gamma": 0.95, + "inertia": 0.65, + "kappa": 0.0, + "n_background": 30, + "n_minus": 5, + "n_plus": 5, + "perturb_every": 4, + "perturb_sigma": 0.03, + "predictor_eta": 0.2, + "process_noise": 0.12, + "steps_per_episode": 28, + "target": 0.8, + "terminal_reward": 1.0, + "vectorizer_eta": 0.03, + "velocity_reward_scale": 0.25 + }, + "finite": true, + "hardware": { + "device": "cpu", + "platform": "Linux-5.15.0-161-generic-x86_64-with-glibc2.35", + "threads": 1, + "torch_version": "2.10.0+cu128" + }, + "peak_rss_mib": 951.56640625, + "protocol": { + "confirmation_seeds_touched": false, + "d4_gate_sha256": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "grid_size": 24, + "name": "oral_b_v2_development_v1", + "old_r2_gate_sha256": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol_sha256": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "selection_split": "development" + }, + "provenance": { + "git_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "git_tracked_dirty": false, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "tracked_inputs": { + "base_dynamics": true, + "confirmation_analyzer": true, + "confirmation_runner": true, + "d4_gate": true, + "development_analyzer": true, + "old_r2_gate": true, + "protocol": true, + "runner": true, + "v2_dynamics": true, + "v2_metrics": true + } + }, + "schema_version": 2, + "signatures": { + "acute_outcome_lesion_outcome_balanced_acc": 0.5, + "acute_outcome_lesion_role_aligned_separation": -0.0014301145214783086, + "causal_role_sign_inversion_index": 0.007041694604717213, + "challenge_episodes": 896, + "challenge_failure_count": 896, + "challenge_success_count": 0, + "challenge_success_fraction": 0.0, + "critic_contribution_value_prediction_corr": 0.9999999999968772, + "decoder_distance_residual_corr": 0.28710856472496366, + "horizon_success_fraction": { + "12": 0.0, + "16": 0.0, + "20": 0.0, + "24": 0.0, + "28": 0.0, + "4": 0.0, + "8": 0.0 + }, + "mean_abs_raw_soma_corr": 0.9999712737829434, + "mean_abs_residual_soma_corr": 0.036034544915666086, + "mean_critic_expectedness_contribution": -0.0012722876320388449, + "nonterminal_training_events": 24192, + "raw_minus_residual_abs_soma_corr": 0.9639367288672773, + "role_aligned_error_cv_corr": 0.2957875629760282, + "role_aligned_velocity_cv_corr": 0.9990010136460935, + "surrounding_event_decoder_balanced_acc": 0.5914244423899175, + "terminal_outcome_separation_drop_under_acute_lesion": 0.0, + "terminal_previous_soma_outcome_balanced_acc": 0.5, + "terminal_residual_minus_previous_soma_acc": 0.0, + "terminal_residual_outcome_balanced_acc": 0.5, + "terminal_role_aligned_outcome_separation": -0.0014301145214783086, + "terminal_training_events": 896, + "velocity_minus_error_abs_cv_corr": 0.7032134506700654 + }, + "split": "development", + "wall_s": 2.884601291269064, + "warmup": { + "critic_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "fixed_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": -0.1393236517906189, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "intact": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "oracle_role": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 1.0000001192092896, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": false + }, + "outcome_training_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + }, + "plasticity_lesion": { + "batches": 100, + "examples": 6400, + "instruction_present": false, + "predictor_max_abs_error": 2.384185791015625e-07, + "rng_seed": 602200, + "role_cosine_after_warmup": 0.9940935969352722, + "role_cursor_scalar_observations": 12800, + "role_update_enabled": true + } + } +} diff --git a/results/bci_v2_dev_gate.json b/results/bci_v2_dev_gate.json new file mode 100644 index 0000000..c7c3c3c --- /dev/null +++ b/results/bci_v2_dev_gate.json @@ -0,0 +1,701 @@ +{ + "candidate_summaries": [ + { + "checks_by_task_seed": { + "20": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "21": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "22": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + } + }, + "critic_eta": 0.01, + "eligible": false, + "forward_eta": 0.03, + "gamma": 0.8, + "intact_final_by_task_seed": [ + 0.0, + 0.0, + 0.0 + ], + "mean_intact_final": 0.0, + "terminal_outcome_accuracy_by_task_seed": [ + 0.5, + 0.5, + 0.5 + ], + "worst_intact_final": 0.0, + "worst_terminal_outcome_accuracy": 0.5 + }, + { + "checks_by_task_seed": { + "20": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "21": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "22": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + } + }, + "critic_eta": 0.03, + "eligible": false, + "forward_eta": 0.03, + "gamma": 0.8, + "intact_final_by_task_seed": [ + 0.0, + 0.0, + 0.0 + ], + "mean_intact_final": 0.0, + "terminal_outcome_accuracy_by_task_seed": [ + 0.5, + 0.5, + 0.5 + ], + "worst_intact_final": 0.0, + "worst_terminal_outcome_accuracy": 0.5 + }, + { + "checks_by_task_seed": { + "20": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "21": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "22": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + } + }, + "critic_eta": 0.01, + "eligible": false, + "forward_eta": 0.03, + "gamma": 0.95, + "intact_final_by_task_seed": [ + 0.0, + 0.0, + 0.0 + ], + "mean_intact_final": 0.0, + "terminal_outcome_accuracy_by_task_seed": [ + 0.5, + 0.5, + 0.5 + ], + "worst_intact_final": 0.0, + "worst_terminal_outcome_accuracy": 0.5 + }, + { + "checks_by_task_seed": { + "20": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "21": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "22": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + } + }, + "critic_eta": 0.03, + "eligible": false, + "forward_eta": 0.03, + "gamma": 0.95, + "intact_final_by_task_seed": [ + 0.0, + 0.0, + 0.0 + ], + "mean_intact_final": 0.0, + "terminal_outcome_accuracy_by_task_seed": [ + 0.5, + 0.5, + 0.5 + ], + "worst_intact_final": 0.0, + "worst_terminal_outcome_accuracy": 0.5 + }, + { + "checks_by_task_seed": { + "20": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "21": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "22": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + } + }, + "critic_eta": 0.01, + "eligible": false, + "forward_eta": 0.1, + "gamma": 0.8, + "intact_final_by_task_seed": [ + 0.0, + 0.0, + 0.0 + ], + "mean_intact_final": 0.0, + "terminal_outcome_accuracy_by_task_seed": [ + 0.5, + 0.5, + 0.5 + ], + "worst_intact_final": 0.0, + "worst_terminal_outcome_accuracy": 0.5 + }, + { + "checks_by_task_seed": { + "20": { + "acute_outcome_lesion_separation_drop_at_least_0p01": true, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": true, + "terminal_role_separation_at_least_0p03": true, + "velocity_advantage_at_least_0p05": true + }, + "21": { + "acute_outcome_lesion_separation_drop_at_least_0p01": true, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": true, + "velocity_advantage_at_least_0p05": true + }, + "22": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + } + }, + "critic_eta": 0.03, + "eligible": false, + "forward_eta": 0.1, + "gamma": 0.8, + "intact_final_by_task_seed": [ + 0.00390625, + 0.0078125, + 0.0 + ], + "mean_intact_final": 0.00390625, + "terminal_outcome_accuracy_by_task_seed": [ + 1.0, + 0.5, + 0.5 + ], + "worst_intact_final": 0.0, + "worst_terminal_outcome_accuracy": 0.5 + }, + { + "checks_by_task_seed": { + "20": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "21": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "22": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + } + }, + "critic_eta": 0.01, + "eligible": false, + "forward_eta": 0.1, + "gamma": 0.95, + "intact_final_by_task_seed": [ + 0.0, + 0.0, + 0.0 + ], + "mean_intact_final": 0.0, + "terminal_outcome_accuracy_by_task_seed": [ + 0.5, + 0.5, + 0.5 + ], + "worst_intact_final": 0.0, + "worst_terminal_outcome_accuracy": 0.5 + }, + { + "checks_by_task_seed": { + "20": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "21": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + }, + "22": { + "acute_outcome_lesion_separation_drop_at_least_0p01": false, + "challenge_success_fraction_between_0p15_and_0p85": false, + "critic_contribution_tracks_value_at_least_0p95": true, + "critic_expectedness_contribution_at_least_0p005": false, + "decoder_distance_corr_at_least_0p02": true, + "final_success_at_least_0p70": false, + "fixed_role_final_gap_at_least_0p20": false, + "learned_role_cosine_at_least_0p80": true, + "learning_gain_at_least_0p10": false, + "oracle_final_deficit_at_most_0p10": true, + "plasticity_lesion_retains_at_most_half_gain": true, + "raw_residual_corr_gap_at_least_0p20": true, + "residual_soma_corr_at_most_0p10": true, + "sign_inversion_at_least_0p01": false, + "surrounding_event_accuracy_at_least_0p52": true, + "terminal_outcome_accuracy_at_least_0p65": false, + "terminal_role_separation_at_least_0p03": false, + "velocity_advantage_at_least_0p05": true + } + }, + "critic_eta": 0.03, + "eligible": false, + "forward_eta": 0.1, + "gamma": 0.95, + "intact_final_by_task_seed": [ + 0.0, + 0.0, + 0.0 + ], + "mean_intact_final": 0.0, + "terminal_outcome_accuracy_by_task_seed": [ + 0.5, + 0.5, + 0.5 + ], + "worst_intact_final": 0.0, + "worst_terminal_outcome_accuracy": 0.5 + } + ], + "complete_grid": true, + "confirmation_seeds_touched": false, + "grid_size": 24, + "input_sha256": { + "base_dynamics": "d5a373314562af0daedf55baa04b975fe643236c7821be1560d2cfb4359688fc", + "confirmation_analyzer": "e18d2f0cee6ee3be74fdf309b51ab80d114206c517918caa501e24379e2046a6", + "confirmation_runner": "77ee8fa3923481789dba5a864228dd05ef4cc5039ce7f9b4876c3aba477a4939", + "d4_gate": "636c587e47287338cdca7d9558dc3bfb99a2cec615badb23337025d85b8128ef", + "development_analyzer": "f5fd61493e2264d4cf39f6915a52a265d932703483a6cb6b063228c2c1ee190d", + "old_r2_gate": "4f6f969ceae88afa2523e3472a3373522991f5ecaab69440830c07479d2d3597", + "protocol": "940bac74b527605d33a36a2b79f404a69de497a07d7657e5341ad28c4f1373f7", + "runner": "4157a57806dedfc2eb70ec84f7a7ca1f73ab684e5d8f0b227cd80465e8e97aa5", + "v2_dynamics": "786c5aed3091644272ae0a740913ac3cc38cdb02a8cacb6dfc8496e36836cdae", + "v2_metrics": "421da9287b92e62c742f9eeefbfdccba582c6a4ad5cf92ade4709689120fa1ee" + }, + "oral_b_v2_confirmation_opened": false, + "protocol": "oral_b_v2_development_v1", + "review_score_after": 7, + "review_score_before": 7, + "score_change_rule": "development selection never changes the formal milestone score", + "selected": null, + "source_commit": "670c1399659f31ac79c4b454147417d4f8ddca7d", + "source_sha256": { + "results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p01_t20_m0.json": "143e0e14db46947550164eb9a4e9c2c0a0a1dfdbfe86801c1a1d71392fee16c3", + "results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p01_t21_m0.json": "629825d2c0214efa7292b343456375e31f21e71582471d24f97287c971e3a5ce", + "results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p01_t22_m0.json": "e96062abf060175e3d432df7bd784802797238d09a28e373d37d5f5b040c951b", + "results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p03_t20_m0.json": "a69c813e54b27466cbfa78aa1a379186d3f693fa51f06449d87c4319da9f4ee6", + "results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p03_t21_m0.json": "428c0d5f32ca340af33e5a0c87390759b965d8ffa5cb322ddd484c25e7f31a04", + "results/bci_v2_dev/bci_v2_e0p03_g0p8_c0p03_t22_m0.json": "0d0fc07a7ca61905377547372dd3fd719cb3187a0e75fbb48bba43cc7362d1c0", + "results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p01_t20_m0.json": "e63b6d190e7c62fe5e949739033b5f495fd9f7275aac8f0f157a8511eebf45a4", + "results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p01_t21_m0.json": "c0bb7e92b786f044733c5e3d639b07186cadbb786681783cd5fd38794f35417d", + "results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p01_t22_m0.json": "a32e1935cb2c4677d66a613e431ea6943f2a759e0a0c43eb55e930c73804f2cd", + "results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p03_t20_m0.json": "5fa8d08c22f76261a3356f4e08bd544818359a2c6dfb42e4f0e1c5c3fb543ef7", + "results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p03_t21_m0.json": "7f6d0e865e4e6b74b701ea754b6f3c1610df0fd4a1948e38b8dd6ca497beb647", + "results/bci_v2_dev/bci_v2_e0p03_g0p95_c0p03_t22_m0.json": "76a88a2cc141e454d1e0c8ea355a0c719e7afba6c16d39e3695da28f7ba4b90d", + "results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p01_t20_m0.json": "2d6f62aa84735a1708e969520bf4a97c2fc5679776a9d20d413ac4dfdcd75863", + "results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p01_t21_m0.json": "b7d3563d491b1218ceeb70d0a70fb03cea333628919c02dbf27f58a403c9232f", + "results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p01_t22_m0.json": "789da0c36f6925716a7d17d00321164c8e6a398af9e9636d7b341076945c1e6d", + "results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p03_t20_m0.json": "eea173c05c402903a85fbc0b848a8f970fbd4dbe7af8bd0e88e0c3c7ebefc9b6", + "results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p03_t21_m0.json": "443369d0bedcefd90a608f54bcf0983eab4a426d8467b0de1e20e0d30360d131", + "results/bci_v2_dev/bci_v2_e0p1_g0p8_c0p03_t22_m0.json": "f5f4ac51802bbe8fa0818a5500150f4bee26f8b986c9824c830a0cc9641834bf", + "results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p01_t20_m0.json": "00bd62c71f93c8cf456bcee62b33a97cbe4183722f48b122cb6f0dbe36e8b0fa", + "results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p01_t21_m0.json": "b14a09750d267be9d0aab0560f19a1eb590e4b54e7843d08cdc49b8f086e2f93", + "results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p01_t22_m0.json": "af00424b8a3a447c4d55726b5bbbb03782a7df70eb382c12b5f9212e499de197", + "results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p03_t20_m0.json": "d4ac8619a67e10ef23d1d6c3e70334314a093b9edca5c1033cd30e316dfddab1", + "results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p03_t21_m0.json": "9776a1d977fa613025c024c13a26a73a64b2b0f4578de954e4020b692eff5e18", + "results/bci_v2_dev/bci_v2_e0p1_g0p95_c0p03_t22_m0.json": "b798dab06ae1ddb88b8d8912916d322b4024ca849435bc16150532ee102738e7" + }, + "status": "failed" +} |
