summaryrefslogtreecommitdiff
path: root/external/dualprop_patches/0019-experiment-add-portable-A6000-P2-profile.patch
diff options
context:
space:
mode:
Diffstat (limited to 'external/dualprop_patches/0019-experiment-add-portable-A6000-P2-profile.patch')
-rw-r--r--external/dualprop_patches/0019-experiment-add-portable-A6000-P2-profile.patch217
1 files changed, 217 insertions, 0 deletions
diff --git a/external/dualprop_patches/0019-experiment-add-portable-A6000-P2-profile.patch b/external/dualprop_patches/0019-experiment-add-portable-A6000-P2-profile.patch
new file mode 100644
index 0000000..d5e0542
--- /dev/null
+++ b/external/dualprop_patches/0019-experiment-add-portable-A6000-P2-profile.patch
@@ -0,0 +1,217 @@
+From c88ebbc422cb7ff8ae117d924e6fa28c41729022 Mon Sep 17 00:00:00 2001
+From: YurenHao0426 <Blackhao0426@gmail.com>
+Date: Thu, 30 Jul 2026 16:31:36 -0500
+Subject: [PATCH 19/19] experiment: add portable A6000 P2 profile
+
+---
+ analyze_p2.py | 16 +++++++++--
+ crossover_grid.py | 71 +++++++++++++++++++++++++++++++++++++++--------
+ 2 files changed, 74 insertions(+), 13 deletions(-)
+
+diff --git a/analyze_p2.py b/analyze_p2.py
+index d997ec5..696fef0 100644
+--- a/analyze_p2.py
++++ b/analyze_p2.py
+@@ -137,8 +137,18 @@ def audit_record(job, launch):
+ assert manifest["selector_sha256"] == crossover_grid.SELECTOR_SHA256
+ assert manifest["work"] == crossover_grid.p2_work_report(job)
+ hardware = manifest["hardware"]
+- assert hardware["physical_index"] in (5, 7)
+- assert hardware["name"] == "NVIDIA GeForce GTX 1080"
++ hardware_policy = launch["hardware_policy"]
++ assert hardware["hardware_profile"] == hardware_policy["profile"]
++ assert (
++ hardware["physical_index"]
++ in hardware_policy["allowed_physical_gpu_indices"]
++ )
++ assert (
++ hardware["cuda_visible_devices"]
++ == str(hardware["physical_index"])
++ )
++ assert hardware["uuid"]
++ assert hardware["name"] in hardware_policy["allowed_device_names"]
+ assert hardware["peak_process_gpu_memory_mib"] >= 0
+ histogram = manifest["histogram"]
+ if histogram is None:
+@@ -188,6 +198,8 @@ def main():
+ jobs)
+ assert launch["jobs"] == jobs
+ assert launch["selector_sha256"] == crossover_grid.SELECTOR_SHA256
++ assert launch["hardware_policy"]["profile"] in (
++ "timan107_gtx1080", "a6000")
+ assert launch["source"]["cifar10_tfds_tree"]["num_files"] == 5
+ expected = {job["experiment_name"]: job for job in jobs}
+ actual = {
+diff --git a/crossover_grid.py b/crossover_grid.py
+index bb6d9d5..1f14d21 100644
+--- a/crossover_grid.py
++++ b/crossover_grid.py
+@@ -12,7 +12,7 @@ import time
+
+
+ ROOT = os.path.dirname(os.path.abspath(__file__))
+-MAIN_ROOT = "/home/yurenh2/sdil"
++MAIN_ROOT = os.environ.get("SDIL_MAIN_ROOT", "/home/yurenh2/sdil")
+ PROTOCOL = os.path.join(MAIN_ROOT, "PLAIN_CNN_CROSSOVER.md")
+ SELECTOR = os.path.join(
+ MAIN_ROOT, "results", "plain_cnn_p1_selector.json")
+@@ -22,6 +22,16 @@ P2_METHODS = (
+ "bp", "fa", "dfa", "pepita", "ff", "ep", "dualprop", "clean_kp",
+ "sdil")
+ P2_ARCHITECTURES = ("minicnn", "vgglike", "vgg16")
++HARDWARE_PROFILES = {
++ "timan107_gtx1080": {
++ "allowed_physical_gpu_indices": [5, 7],
++ "allowed_device_names": ["NVIDIA GeForce GTX 1080"],
++ },
++ "a6000": {
++ "allowed_physical_gpu_indices": None,
++ "allowed_device_names": ["NVIDIA RTX A6000"],
++ },
++}
+
+
+ def output(*args):
+@@ -185,7 +195,7 @@ def p2_registry_sha256(jobs):
+ return hashlib.sha256(encoded).hexdigest()
+
+
+-def ensure_p2_launch(source, jobs):
++def ensure_p2_launch(source, jobs, hardware_policy):
+ path = os.path.join(ROOT, "runs", "plain-p2-launch.json")
+ immutable = {
+ "stage": "p2",
+@@ -194,6 +204,7 @@ def ensure_p2_launch(source, jobs):
+ "selector_sha256": SELECTOR_SHA256,
+ "num_registered_cells": len(jobs),
+ "registry_sha256": p2_registry_sha256(jobs),
++ "hardware_policy": hardware_policy,
+ "jobs": jobs,
+ }
+ if os.path.exists(path):
+@@ -413,11 +424,22 @@ def completed_histogram(experiment_name):
+ return paths[0]
+
+
+-def physical_gpu_report():
++def physical_gpu_report(profile):
+ index = os.environ.get("CUDA_VISIBLE_DEVICES")
+- if index not in ("5", "7"):
++ if index is None or not index.isdigit():
+ raise RuntimeError(
+- "formal P2 is authorized only on physical GPUs 5 and 7")
++ "formal P2 requires CUDA_VISIBLE_DEVICES to contain one "
++ "physical numeric GPU index")
++ if profile not in HARDWARE_PROFILES:
++ raise ValueError(f"unknown hardware profile: {profile}")
++ configured = HARDWARE_PROFILES[profile]
++ allowed_indices = configured["allowed_physical_gpu_indices"]
++ if (
++ allowed_indices is not None
++ and int(index) not in allowed_indices
++ ):
++ raise RuntimeError(
++ f"physical GPU {index} is not allowed by {profile}")
+ query = output(
+ "nvidia-smi",
+ "--query-gpu=index,uuid,name,memory.total",
+@@ -427,7 +449,13 @@ def physical_gpu_report():
+ if len(matches) != 1:
+ raise RuntimeError(f"cannot resolve physical GPU {index}")
+ physical_index, uuid, name, memory_mib = matches[0]
++ if name not in configured["allowed_device_names"]:
++ raise RuntimeError(
++ f"{profile} requires {configured['allowed_device_names']}, "
++ f"found {name}")
+ return {
++ "hardware_profile": profile,
++ "cuda_visible_devices": index,
+ "physical_index": int(physical_index),
+ "uuid": uuid,
+ "name": name,
+@@ -435,6 +463,19 @@ def physical_gpu_report():
+ }
+
+
++def hardware_policy(profile, gpu):
++ configured = HARDWARE_PROFILES[profile]
++ allowed_indices = configured["allowed_physical_gpu_indices"]
++ if allowed_indices is None:
++ allowed_indices = [gpu["physical_index"]]
++ return {
++ "profile": profile,
++ "allowed_physical_gpu_indices": list(allowed_indices),
++ "allowed_device_names": list(configured["allowed_device_names"]),
++ "single_gpu_per_process": True,
++ }
++
++
+ def descendant_pids(root_pid):
+ listing = subprocess.run(
+ ["ps", "-eo", "pid=,ppid="], check=True, capture_output=True,
+@@ -597,7 +638,7 @@ def terminate_process_group(process):
+ process.wait()
+
+
+-def run_p2_job(job, source, dry_run):
++def run_p2_job(job, source, gpu, dry_run):
+ existing = record_path(job["experiment_name"])
+ if existing is not None:
+ print(f"preserving {job['experiment_name']} -> {existing}", flush=True)
+@@ -609,7 +650,6 @@ def run_p2_job(job, source, dry_run):
+ print("RUN", " ".join(job["command"]), flush=True)
+ if dry_run:
+ return
+- gpu = physical_gpu_report()
+ started = time.time()
+ process = subprocess.Popen(
+ job["command"], cwd=ROOT, start_new_session=True)
+@@ -666,9 +706,9 @@ def run_p2_job(job, source, dry_run):
+ flush=True)
+
+
+-def run_job(job, source, dry_run):
++def run_job(job, source, gpu, dry_run):
+ if job["stage"] == "p2":
+- return run_p2_job(job, source, dry_run)
++ return run_p2_job(job, source, gpu, dry_run)
+ result = completed_histogram(job["experiment_name"])
+ if result is not None:
+ print(f"preserving {job['experiment_name']} -> {result}", flush=True)
+@@ -703,14 +743,23 @@ def main():
+ parser.add_argument("--shard-index", type=int, default=0)
+ parser.add_argument("--num-shards", type=int, default=1)
+ parser.add_argument("--method")
++ parser.add_argument(
++ "--hardware-profile",
++ choices=tuple(HARDWARE_PROFILES),
++ default=os.environ.get(
++ "SDIL_HARDWARE_PROFILE", "timan107_gtx1080"),
++ )
+ parser.add_argument("--dry-run", action="store_true")
+ args = parser.parse_args()
+ if not 0 <= args.shard_index < args.num_shards:
+ raise ValueError("invalid shard")
+ source = source_report()
+ jobs = p1_jobs() if args.stage == "p1" else p2_jobs()
++ gpu = None
+ if args.stage == "p2" and not args.dry_run:
+- launch = ensure_p2_launch(source, jobs)
++ gpu = physical_gpu_report(args.hardware_profile)
++ launch = ensure_p2_launch(
++ source, jobs, hardware_policy(args.hardware_profile, gpu))
+ print(f"P2 launch lock: {launch}", flush=True)
+ if args.method:
+ jobs = [job for job in jobs if job["method"] == args.method]
+@@ -721,7 +770,7 @@ def main():
+ if not jobs:
+ raise ValueError("no jobs selected")
+ for job in jobs:
+- run_job(job, source, args.dry_run)
++ run_job(job, source, gpu, args.dry_run)
+
+
+ if __name__ == "__main__":
+--
+2.54.0
+