Files
foxhunt/scripts/compare-nsys-profiles.py
jgrusewski 2606506cd8 plan5(task3): A.4.1 nsys profile harness with regression-comparison script
- argo-train.sh: --profile flag forces multi-seed render path so the
  nsys wrapper + foxhunt-training-artifacts upload step are visible in
  --dry-run YAML without cluster contact (test surface).
- train-multi-seed-template.yaml: new `profile` parameter (default
  "false") gates the per-(seed, fold) `nsys profile
  --capture-range=cudaProfilerApi` wrapper and the `mc cp` upload to
  foxhunt-training-artifacts/profiles/<sha>/. mc binary fetched
  on-demand (ci-builder image lacks it). MinIO creds optional —
  upload warn-skips if absent.
- Dockerfile.foxhunt-training-runtime: install nsight-systems-cli
  unpinned (pinning the stale 2024.4.1.61-1 from earlier plans
  breaks builds when apt index advances).
- minio.yaml: add foxhunt-training-artifacts bucket to minio-init.
- compare-nsys-profiles.py: V0 regression detector — compares
  cuda_gpu_kern_sum total_ns / epoch_count between two profiles;
  exits 1 on >20% slowdown. NVTX per-epoch ranges deferred to T5.
- tests/test_nsys_harness.sh: dry-run grep test — verifies both
  required strings appear when --profile is set, and that the
  default (no --profile) path keeps profile=false in the rendered
  template.
- dqn-wire-up-audit.md: Plan 5 Task 3 row added documenting the
  harness + the baseline-capture deferral to T5.

Backward compat: test_multi_seed_harness.sh from P5T1 still PASS.
2026-04-26 12:25:35 +02:00

167 lines
5.4 KiB
Python
Executable File

#!/usr/bin/env python3
"""Compare two nsys .nsys-rep profiles for >20% per-epoch wall-time regression.
Plan 5 Task 3 (A.4.1) — perf-regression detector.
V0 implementation: total GPU wall-time / epoch_count. The training loop does
not yet emit explicit NVTX `epoch=N` ranges that nsys could bucket on, so we
treat the median of (total_gpu_time / num_epochs) as the per-epoch estimate.
This is a coarser measure than per-epoch ranges but is robust against trace
parsing edge cases and is the same comparison the multi-seed wall-time
sanity check would emit. A V1 follow-up (Plan 5 Task 5) can add NVTX
push/pop ranges in `training_loop.rs` and switch this script to
`nsys stats --report nvtx_pushpop_trace` for true per-epoch granularity.
Usage:
compare-nsys-profiles.py BASELINE.nsys-rep CURRENT.nsys-rep [--epochs N]
Exit codes:
0 — within 20% of baseline (PASS)
1 — >20% per-epoch regression (FAIL)
2 — invalid input / nsys failure (ERROR)
"""
from __future__ import annotations
import argparse
import csv
import io
import sys
from pathlib import Path
from subprocess import CalledProcessError, run
REGRESSION_THRESHOLD = 1.20 # 20% slowdown trips FAIL.
def extract_total_gpu_time_ms(nsys_rep: Path) -> float:
"""Return total GPU kernel wall-time in ms across the entire trace.
Parses `nsys stats --report cuda_gpu_kern_sum --format csv` and sums the
Total Time column. nsys emits time in ns by default; we normalise to ms
so the comparison is invariant to nsys's display-unit quirks.
"""
if not nsys_rep.exists():
raise FileNotFoundError(f"nsys-rep not found: {nsys_rep}")
try:
proc = run(
[
"nsys",
"stats",
"--report",
"cuda_gpu_kern_sum",
"--format",
"csv",
str(nsys_rep),
],
capture_output=True,
text=True,
check=True,
)
except FileNotFoundError as exc:
raise RuntimeError(
"nsys binary not found in PATH — install nsight-systems-cli"
) from exc
except CalledProcessError as exc:
raise RuntimeError(
f"nsys stats failed for {nsys_rep}: {exc.stderr.strip()}"
) from exc
# nsys emits header + comment lines before the CSV. Find the header row
# ("Time (%),Total Time...") and parse the rest.
csv_lines: list[str] = []
seen_header = False
for line in proc.stdout.splitlines():
if not seen_header:
if line.startswith('"Time (%)"') or line.startswith("Time (%)"):
seen_header = True
csv_lines.append(line)
continue
csv_lines.append(line)
if not csv_lines:
raise RuntimeError(
f"could not locate CSV header in nsys stats output for {nsys_rep}"
)
reader = csv.DictReader(io.StringIO("\n".join(csv_lines)))
total_ns = 0.0
rows_parsed = 0
for row in reader:
# Column key tolerates both "Total Time" and "Total Time (ns)" forms
# across nsys versions.
for key in row:
if key and key.startswith("Total Time"):
try:
total_ns += float(row[key].replace(",", ""))
rows_parsed += 1
except (ValueError, AttributeError):
continue
break
if rows_parsed == 0:
raise RuntimeError(
f"no Total Time rows parsed from {nsys_rep} — empty trace?"
)
return total_ns / 1_000_000.0 # ns -> ms
def per_epoch_ms(nsys_rep: Path, epochs: int) -> float:
"""Per-epoch wall-time estimate (V0: total / epochs)."""
if epochs <= 0:
raise ValueError(f"epochs must be > 0, got {epochs}")
return extract_total_gpu_time_ms(nsys_rep) / epochs
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(
description=(
"Compare per-epoch GPU wall-time between two nsys profiles; "
"fail on >20% regression."
)
)
parser.add_argument("baseline", type=Path, help="Baseline .nsys-rep path")
parser.add_argument("current", type=Path, help="Current .nsys-rep path")
parser.add_argument(
"--epochs",
type=int,
default=5,
help=(
"Epoch count both profiles ran (default 5 — matches "
"multi_fold_convergence smoke). Both runs MUST use the same "
"epoch count — comparing different epoch budgets is meaningless."
),
)
args = parser.parse_args(argv)
try:
b_per_epoch = per_epoch_ms(args.baseline, args.epochs)
c_per_epoch = per_epoch_ms(args.current, args.epochs)
except (FileNotFoundError, RuntimeError, ValueError) as exc:
print(f"ERROR: {exc}", file=sys.stderr)
return 2
ratio = c_per_epoch / b_per_epoch if b_per_epoch > 0 else float("inf")
summary = (
f"baseline={b_per_epoch:.1f}ms/epoch "
f"current={c_per_epoch:.1f}ms/epoch "
f"ratio={ratio:.3f}x "
f"(epochs={args.epochs})"
)
if ratio > REGRESSION_THRESHOLD:
print(
f"FAIL: per-epoch regression {ratio:.2f}x exceeds "
f"{REGRESSION_THRESHOLD:.2f}x threshold | {summary}"
)
return 1
print(f"PASS: per-epoch within {REGRESSION_THRESHOLD:.2f}x | {summary}")
return 0
if __name__ == "__main__":
sys.exit(main())