summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorYurenHao0426 <Blackhao0426@gmail.com>2026-07-22 14:31:04 -0500
committerYurenHao0426 <Blackhao0426@gmail.com>2026-07-22 14:33:51 -0500
commit73a7e1a0476fa7b21353afb8d9077d66e48d5639 (patch)
treeacb6f806b714b309bd5335d4c8458a97de650655
parent39d1839ead0709798c2eb07104c3764ccf1ed3ae (diff)
protocol: implement reciprocal KP full gate
-rwxr-xr-xexperiments/analyze_kp_full.py128
-rwxr-xr-xexperiments/kp_full_development.py44
2 files changed, 172 insertions, 0 deletions
diff --git a/experiments/analyze_kp_full.py b/experiments/analyze_kp_full.py
new file mode 100755
index 0000000..024eb10
--- /dev/null
+++ b/experiments/analyze_kp_full.py
@@ -0,0 +1,128 @@
+#!/usr/bin/env python3
+"""Audit and gate the frozen KP-2 full ResNet-20 validation baseline."""
+import argparse
+import json
+import math
+import os
+
+
+SPLIT_HASH = "8328b206a97c420e49e54e3eca4abe3274c4756b084355784ea3fb8059e4515b"
+
+
+def main():
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--selection", default="results/kp_short_gate.json")
+ parser.add_argument(
+ "--bp_selection", default="results/oral_a_bp_selection.json")
+ parser.add_argument("--input", default="results/kp_full/kp.json")
+ parser.add_argument("--out", default="results/kp_full_gate.json")
+ args = parser.parse_args()
+ with open(args.selection) as handle:
+ selection = json.load(handle)
+ if selection.get("protocol") != "kolen_pollack_short_v1":
+ raise ValueError("unexpected KP-1 selection protocol")
+ if selection.get("status") != "passed":
+ raise ValueError("KP-1 did not open KP-2")
+ with open(args.input) as handle:
+ record = json.load(handle)
+ run_args = record["args"]
+ expected = {
+ "mode": "kp", "depth": 20, "width": 16, "seed": 0,
+ "loader_seed": 0, "batch_size": 128, "epochs": 200,
+ "train_limit": 0, "val_examples": 5000, "split_seed": 2027,
+ "eval_split": "validation", "eval_every": 0,
+ "augment_train": 1, "lr": 0.1, "output_lr": 0.1,
+ "lr_schedule": "step", "lr_milestones": "100,150",
+ "lr_gamma": 0.1, "warmup_epochs": 0, "momentum": 0.9,
+ "weight_decay": 1e-4, "normalization": "batchnorm",
+ "a_scale": 1.0, "alignment_probe": 32,
+ }
+ for key, value in expected.items():
+ if run_args.get(key) != value:
+ raise ValueError(f"KP-2 {key} drift")
+ if record["provenance"]["git_tracked_dirty"]:
+ raise ValueError("tracked-dirty KP-2 result")
+ if record["split"]["validation_index_sha256"] != SPLIT_HASH:
+ raise ValueError("KP-2 split drift")
+ protocol = record["evaluation_protocol"]
+ if protocol["test_evaluations"] or protocol["test_used_for_selection"]:
+ raise ValueError("KP-2 touched test")
+ if record.get("calibration_metric_space") != (
+ "reciprocal_local_activity_products"):
+ raise ValueError("KP-2 metric-space drift")
+
+ with open(args.bp_selection) as handle:
+ bp_selection = json.load(handle)
+ if bp_selection.get("status") != "passed_primary":
+ raise ValueError("matched full BP reference is not frozen")
+ with open(bp_selection["selected"]["path"]) as handle:
+ bp = json.load(handle)
+ bp_accuracy = float(bp["final"]["accuracy"])
+ bp_macs = int(bp["work"]["total_macs_estimate"])
+
+ tracking = [row.get("feedback_tracking") for row in record["epochs"]]
+ if len(tracking) != 200 or any(value is None for value in tracking):
+ raise ValueError("KP-2 tracking trajectory is incomplete")
+ trajectory_values = []
+ for row, values in zip(record["epochs"], tracking):
+ trajectory_values.extend([
+ float(row["train_loss"]),
+ float(values["mean_feedback_forward_cosine"]),
+ float(values["mean_feedback_forward_relative_error"]),
+ float(values["min_feedback_forward_cosine"]),
+ float(values["max_feedback_forward_relative_error"]),
+ ])
+ diagnostics = record["diagnostics"]
+ accuracy = float(record["final"]["accuracy"])
+ loss = float(record["final"]["loss"])
+ early = float(diagnostics["early_third_mean"])
+ final_cosine = float(diagnostics["mean_feedback_forward_cosine"])
+ late_cosine = sum(float(value["mean_feedback_forward_cosine"])
+ for value in tracking[150:]) / 50
+ total_macs = int(record["work"]["total_macs_estimate"])
+ queries = int(record["work"]["logical_batch_loss_queries"])
+ finite = (bool(record["final"]["finite"])
+ and all(math.isfinite(value) for value in
+ trajectory_values + [accuracy, loss, early, final_cosine]))
+ checks = {
+ "record_and_trajectory_finite": finite,
+ "accuracy_at_least_0.88": accuracy >= 0.88,
+ "early_alignment_at_least_0.80": early >= 0.80,
+ "final_feedback_cosine_at_least_0.95": final_cosine >= 0.95,
+ "epoch151_to200_feedback_cosine_at_least_0.95": late_cosine >= 0.95,
+ "zero_task_loss_queries": queries == 0,
+ "macs_at_most_1.40x_bp": total_macs <= 1.40 * bp_macs,
+ }
+ status = "passed" if all(checks.values()) else "failed"
+ metrics = {
+ "accuracy": accuracy, "loss": loss, "bp_accuracy": bp_accuracy,
+ "early_third_alignment": early,
+ "final_mean_feedback_forward_cosine": final_cosine,
+ "epoch151_to200_mean_feedback_forward_cosine": late_cosine,
+ "final_mean_feedback_forward_relative_error": float(
+ diagnostics["mean_feedback_forward_relative_error"]),
+ "total_macs": total_macs, "bp_total_macs": bp_macs,
+ "mac_ratio_to_bp": total_macs / bp_macs,
+ "logical_batch_loss_queries": queries,
+ "peak_memory_allocated_bytes": int(
+ record["hardware"]["peak_memory_allocated_bytes"]),
+ "wall_s": float(record["timing"]["total_timed_wall_s"]),
+ "source_commit": record["provenance"]["git_commit"],
+ }
+ output = {
+ "protocol": "kolen_pollack_full_v1", "status": status,
+ "checks": checks, "metrics": metrics,
+ "innovation_experiment_opened": status == "passed",
+ "confirmation_test_seeds_touched": False,
+ "review_score_before": 5, "review_score_after": 5,
+ "score_change_rule": "inherited KP baseline cannot raise score",
+ }
+ os.makedirs(os.path.dirname(os.path.abspath(args.out)), exist_ok=True)
+ with open(args.out, "w") as handle:
+ json.dump(output, handle, indent=2, sort_keys=True)
+ handle.write("\n")
+ print(json.dumps(output, indent=2))
+
+
+if __name__ == "__main__":
+ main()
diff --git a/experiments/kp_full_development.py b/experiments/kp_full_development.py
new file mode 100755
index 0000000..5378271
--- /dev/null
+++ b/experiments/kp_full_development.py
@@ -0,0 +1,44 @@
+#!/usr/bin/env python3
+"""Run the single frozen KP-2 full ResNet-20 validation baseline."""
+import argparse
+import json
+import os
+import subprocess
+import sys
+
+
+def main():
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--selection", default="results/kp_short_gate.json")
+ parser.add_argument("--device", default="cuda")
+ parser.add_argument("--dry_run", action="store_true")
+ args = parser.parse_args()
+ with open(args.selection) as handle:
+ selection = json.load(handle)
+ if selection.get("protocol") != "kolen_pollack_short_v1":
+ raise ValueError("unexpected KP-1 selection protocol")
+ if selection.get("status") != "passed":
+ raise ValueError("KP-1 did not open KP-2")
+ command = [
+ sys.executable, "experiments/conv_run.py", "--mode", "kp",
+ "--device", args.device, "--depth", "20", "--width", "16",
+ "--seed", "0", "--loader_seed", "0", "--batch_size", "128",
+ "--epochs", "200", "--train_limit", "0",
+ "--val_examples", "5000", "--split_seed", "2027",
+ "--eval_split", "validation", "--eval_every", "0",
+ "--augment_train", "1", "--lr", "0.1", "--output_lr", "0.1",
+ "--lr_schedule", "step", "--lr_milestones", "100,150",
+ "--lr_gamma", "0.1", "--warmup_epochs", "0",
+ "--momentum", "0.9", "--weight_decay", "1e-4",
+ "--normalization", "batchnorm", "--a_scale", "1",
+ "--alignment_probe", "32",
+ "--out", "results/kp_full/kp.json",
+ ]
+ os.makedirs("results/kp_full", exist_ok=True)
+ print(" ".join(command), flush=True)
+ if not args.dry_run:
+ subprocess.run(command, check=True)
+
+
+if __name__ == "__main__":
+ main()