Files
nbrightandClaude Opus 5 609d9a6972 Add SUM Parts reproduction and Seosan Myeongcheon application pipeline
Reproduces the SUM Parts (CVPR 2025) face-labeling benchmark on a single
consumer GPU, then applies it to drone-photogrammetry road survey meshes.

Verified on RTX 3060 12GB / WSL2 Ubuntu 22.04 / CUDA 11.8 / torch 2.0.1:
- CUDA extensions build (pointnet2_batch, pointops, chamfer_dist, emd,
  subsampling)
- PointNet 100 epochs reaches mIoU 17.19, matching the paper's reported 15.1
- OBJ -> PLY conversion round-trips through the model and yields per-point
  predictions

Four upstream source patches, all idempotent, originals preserved:
- numpy aliases removed in 1.24 (np.long etc.) and collections ABCs moved in
  python 3.10
- the blind test split ships label = -1, which crashed ConfusionMatrix
- mode=val referenced `epoch` before assignment

Documents the traps that cost the most time, including VRAM overflow silently
falling back to host RAM on WSL2 (25-100x slowdown, no OOM) and the colour
scale mismatch between r/g/b float32 and red/green/blue uint8.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-21 10:29:25 +09:00

158 lines
5.8 KiB
Python

#!/usr/bin/env python3
"""Measure peak VRAM and per-iteration time for each sumv2_triangle model.
Answers "will full training fit on this GPU, and how long would it take" with
numbers off the actual card rather than a guess. Builds the model and the real
train loader from each cfg, runs a handful of train steps at the cfg's own
batch_size, and reports peak allocated memory plus iterations/second.
Run from PointNeXt_bundle/examples/segmentation.
Usage:
python bench_models.py [--iters 8] [--cfgs pointnet pointnext-xl ...]
"""
from __future__ import annotations
import argparse
import sys
import time
from pathlib import Path
import torch
sys.path.append(str(Path(__file__).resolve().parent))
sys.path.append("../../")
from openpoints.utils import EasyConfig # noqa: E402
from openpoints.dataset import build_dataloader_from_cfg, get_features_by_keys # noqa: E402
from openpoints.models import build_model_from_cfg # noqa: E402
from openpoints.loss import build_criterion_from_cfg # noqa: E402
from openpoints.optim import build_optimizer_from_cfg # noqa: E402
ALL_CFGS = ["pointnet", "pointnet++msg", "pointnext-xl", "pointvector-xl"]
def bench(cfg_name: str, iters: int, voxel_max: int | None = None,
batch_size: int | None = None) -> dict:
cfg = EasyConfig()
cfg.load(f"../../cfgs/sumv2_triangle/{cfg_name}.yaml", recursive=True)
cfg.rank, cfg.distributed, cfg.mp = 0, False, False
if voxel_max is not None:
cfg.dataset.train.voxel_max = voxel_max
if batch_size is not None:
cfg.batch_size = batch_size
model = build_model_from_cfg(cfg.model).cuda()
n_params = sum(p.numel() for p in model.parameters())
cfg.criterion_args.weight = None
criterion = build_criterion_from_cfg(cfg.criterion_args).cuda()
optimizer = build_optimizer_from_cfg(model, lr=cfg.lr, **cfg.optimizer)
train_loader = build_dataloader_from_cfg(
cfg.batch_size, cfg.dataset, cfg.dataloader,
datatransforms_cfg=cfg.datatransforms, split="train", distributed=False,
)
torch.cuda.empty_cache()
torch.cuda.reset_peak_memory_stats()
model.train()
times: list[float] = []
it = iter(train_loader)
n_points = None
for i in range(iters):
try:
data = next(it)
except StopIteration:
it = iter(train_loader)
data = next(it)
for k in data:
data[k] = data[k].cuda(non_blocking=True)
target = data["y"].squeeze(-1)
data["x"] = get_features_by_keys(data, cfg.feature_keys)
if n_points is None:
n_points = int(data["pos"].shape[0] * data["pos"].shape[1]) \
if data["pos"].dim() == 3 else int(data["pos"].shape[0])
torch.cuda.synchronize()
t0 = time.perf_counter()
logits = model(data)
loss = criterion(logits, target)
loss.backward()
optimizer.step()
optimizer.zero_grad()
torch.cuda.synchronize()
dt = time.perf_counter() - t0
if i >= 2: # skip warm-up iterations
times.append(dt)
peak = torch.cuda.max_memory_allocated() / 1024**3
reserved = torch.cuda.max_memory_reserved() / 1024**3
avg = sum(times) / len(times) if times else float("nan")
del model, optimizer, criterion, train_loader
torch.cuda.empty_cache()
return {
"cfg": cfg_name, "params_m": n_params / 1e6, "batch_size": cfg.batch_size,
"voxel_max": cfg.dataset.train.voxel_max, "points_per_batch": n_points,
"peak_gb": peak, "reserved_gb": reserved, "sec_per_iter": avg,
}
def main() -> None:
ap = argparse.ArgumentParser()
ap.add_argument("--iters", type=int, default=8)
ap.add_argument("--cfgs", nargs="*", default=ALL_CFGS)
ap.add_argument("--voxel-max", type=int, default=None,
help="override dataset.train.voxel_max (cfg default: 64000)")
ap.add_argument("--batch-size", type=int, default=None)
args = ap.parse_args()
total_gb = torch.cuda.get_device_properties(0).total_memory / 1024**3
print(f"GPU: {torch.cuda.get_device_name(0)} {total_gb:.1f} GB")
if args.voxel_max or args.batch_size:
print(f"overrides: voxel_max={args.voxel_max} batch_size={args.batch_size}")
print("NOTE: on WSL2 the NVIDIA driver spills past VRAM into host RAM instead")
print(" of raising OOM. A peak above the card's capacity means the run")
print(" was paging over PCIe -- it completes, but uselessly slowly.\n")
rows = []
for name in args.cfgs:
print(f"--- benchmarking {name} ---", flush=True)
try:
rows.append(bench(name, args.iters, args.voxel_max, args.batch_size))
print(f" ok\n", flush=True)
except torch.cuda.OutOfMemoryError as e:
print(f" OOM: {str(e)[:120]}\n", flush=True)
rows.append({"cfg": name, "oom": True})
torch.cuda.empty_cache()
except Exception as e: # noqa: BLE001
print(f" FAILED {type(e).__name__}: {str(e)[:200]}\n", flush=True)
rows.append({"cfg": name, "error": f"{type(e).__name__}: {e}"})
torch.cuda.empty_cache()
print()
print(f"{'cfg':<16}{'params':>9}{'bs':>4}{'pts/batch':>12}"
f"{'peak VRAM':>11}{'s/iter':>9} fits?")
print("-" * 70)
for r in rows:
if r.get("oom"):
print(f"{r['cfg']:<16}{'':>9}{'':>4}{'':>12}{'OOM':>11}{'':>9}")
elif r.get("error"):
print(f"{r['cfg']:<16} {r['error'][:44]}")
else:
fits = "yes" if r["peak_gb"] < total_gb * 0.95 else "NO (spilling)"
print(f"{r['cfg']:<16}{r['params_m']:>8.1f}M{r['batch_size']:>4}"
f"{r['points_per_batch']:>12,}{r['peak_gb']:>10.2f}G"
f"{r['sec_per_iter']:>9.3f} {fits}")
if __name__ == "__main__":
main()