{
  "$schema": "../schemas/collection.schema.json",
  "id": "gpu-mpi-performance",
  "name": "GPU, MPI, And Performance",
  "status": "draft",
  "summary": "Skills for validating GPU allocations, GPU binding, GPU memory failures, TensorBoard training monitors, Ray, JAX, Hugging Face Accelerate, TensorFlow multi-worker training, NCCL communication, DeepSpeed and PyTorch DDP launches, MPI fabric evidence, MPI rank binding, hybrid MPI/OpenMP layouts, BLAS/OpenMP thread pools, CMake build preflight, parallel HDF5/NetCDF preflight, Darshan I/O profile analysis, Lustre striping layout planning, containerized MPI, and mpi4py launches, OpenMP placement, Slurm efficiency review, storage smoke benchmarks, and first-pass performance evidence.",
  "audience": ["AI/HPC users", "simulation teams", "performance engineers"],
  "skill_ids": [
    "mpi-hello-and-benchmark",
    "compiler-mpi-matrix",
    "cmake-hpc-build-preflight",
    "mpi4py-on-slurm",
    "mpi-fabric-diagnostics",
    "mpi-rank-binding-diagnostics",
    "hybrid-mpi-openmp-slurm",
    "blas-openmp-thread-control",
    "parallel-hdf5-netcdf-preflight",
    "darshan-io-profile-analysis",
    "lustre-striping-layout-planning",
    "apptainer-mpi-on-slurm",
    "gpu-sanity-check",
    "slurm-gpu-binding-diagnostics",
    "ray-on-slurm",
    "jax-distributed-on-slurm",
    "huggingface-accelerate-on-slurm",
    "tensorflow-multiworker-on-slurm",
    "pytorch-ddp-on-slurm",
    "nccl-diagnostics",
    "gpu-memory-triage",
    "tensorboard-on-slurm",
    "deepspeed-on-slurm",
    "openmp-thread-affinity",
    "slurm-efficiency-report",
    "ior-mdtest-storage-smoke",
    "performance-profile-basic"
  ],
  "maintainers": [
    {
      "name": "HPC Skill Hub Maintainers"
    }
  ]
}
