From 1fa873de0f662ee8613d7bac73f68aa0a30b033c Mon Sep 17 00:00:00 2001 From: Young Jeong Date: Tue, 15 Sep 2026 23:21:36 -0700 Subject: [PATCH 1/3] B300 noht cmk nccl allreduce test --- cmk-nccltests/README.md | 4 +- cmk-nccltests/nccl-b300-noht.yaml | 141 ++++++++++++++++++++++++++++++ 2 files changed, 144 insertions(+), 1 deletion(-) create mode 100644 cmk-nccltests/nccl-b300-noht.yaml diff --git a/cmk-nccltests/README.md b/cmk-nccltests/README.md index c4c3c94..6cb69c9 100644 --- a/cmk-nccltests/README.md +++ b/cmk-nccltests/README.md @@ -5,6 +5,8 @@ This directory contains Kubernetes manifests for running [NCCL tests](https://gi | File | SKU | GPUs per node | |---|---|---| | [nccl-b200.yaml](nccl-b200.yaml) | B200 180GB SXM | 8 | +| [nccl-b300.yaml](nccl-b300.yaml) | B300 280GB SXM | 8 | +| [nccl-b300-noht.yaml](nccl-b300-nohtt.yaml) | B300 280GB SXM (No Hyperthreading) | 8 | | [nccl-gb200.yaml](nccl-gb200.yaml) | GB200 (NVL72) | 4 | | [nccl-h100.yaml](nccl-h100.yaml) | H100 80GB SXM | 8 | | [nccl-h200.yaml](nccl-h200.yaml) | H200 141GB SXM | 8 | @@ -23,7 +25,7 @@ This directory contains Kubernetes manifests for running [NCCL tests](https://gi 1. Choose the manifest for your SKU. 2. Adjust the `replicas` field under `Worker` to match the number of GPU nodes in your cluster. -3. For B200, H100, and H200, update `-np` in the launcher command to equal ` × ` (e.g. `8 × 2 = 16`). +3. For B300, B200, H100, and H200, update `-np` in the launcher command to equal ` × ` (e.g. `8 × 2 = 16`). 4. For GB200, update `-np` to equal ` × ` (e.g. `4 × 36 = 144`). 5. Apply the manifest: ```bash diff --git a/cmk-nccltests/nccl-b300-noht.yaml b/cmk-nccltests/nccl-b300-noht.yaml new file mode 100644 index 0000000..06dbe16 --- /dev/null +++ b/cmk-nccltests/nccl-b300-noht.yaml @@ -0,0 +1,141 @@ +apiVersion: kubeflow.org/v2beta1 +kind: MPIJob +metadata: + name: nccl-b300-noht +spec: + launcherCreationPolicy: WaitForWorkersReady + slotsPerWorker: 8 + runPolicy: + cleanPodPolicy: Running + mpiReplicaSpecs: + Launcher: + replicas: 1 + template: + spec: + restartPolicy: OnFailure + volumes: + - name: nccl-topo + hostPath: + path: /etc/crusoe/nccl_topo + type: Directory + containers: + - image: ghcr.io/crusoecloud/nccl-tests:13.0.1-ubuntu24.04-nccl-2.29.2-1 + imagePullPolicy: IfNotPresent + volumeMounts: + - name: nccl-topo + mountPath: /opt/nccl_topo + name: nccl-test-launcher + securityContext: + capabilities: + add: ["IPC_LOCK"] + env: + - name: NCCL_TOPO_FILE + value: /opt/nccl_topo/b300-288gb-sxm-ib-noht-cloud-hypervisor.xml + - name: UCX_RNDV_SCHEME + value: "get_zcopy" # UCX memory setting + - name: UCX_TLS + value: "tcp,self" # UCX memory setting + command: + - mpirun + - --allow-run-as-root + - --tag-output + - -np + - "16" # Total number of processes (8 GPUs per node, 2 nodes = 16 total if using 2 Worker Replicas) + - -N + - "8" + - -bind-to + - none + - -map-by + - slot + - -mca + - coll + - ^hcoll + - -x + - NCCL_IB_QPS_PER_CONNECTION=4 + - -x + - NCCL_MIN_CTAS=32 + - -x + - NCCL_MAX_CTAS=32 + - -x + - NCCL_IB_MERGE_VFS=0 + - -x + - NCCL_COLLNET_ENABLE=1 + - -x + - NCCL_IB_HCA=mlx5_5:1,mlx5_6:1,mlx5_7:1,mlx5_8:1,mlx5_9:1,mlx5_10:1,mlx5_11:1,mlx5_12:1 + - -x + - NCCL_TOPO_FILE + - -x + - PATH + - -x + - LD_LIBRARY_PATH + - -x + - NCCL_DEBUG=INFO + - -x + - NCCL_DEBUG_SUBSYS=INIT,NET,ENV + - -x + - NCCL_NVLS_ENABLE=1 + - /opt/nccl-tests/build/all_reduce_perf + - -b + - "2G" + - -e + - "64G" + - -f + - "2" + - -t + - "1" + - -g + - "1" + - -c + - "1" + - -w # Warmup iterations - typically 1, but should be increased for larger clusters + - "50" + - -n + - "100" + Worker: + replicas: 2 # Specify how many worker nodes you have running in the Instances tab + template: + spec: + restartPolicy: OnFailure + runtimeClassName: nvidia + # In case you have GPU taints + tolerations: + - key: "nvidia.com/gpu" + operator: "Exists" + effect: "NoSchedule" + volumes: + - name: dshm + emptyDir: + medium: Memory + sizeLimit: 64Gi + - name: nccl-topo + hostPath: + path: /etc/crusoe/nccl_topo + type: Directory + containers: + - image: ghcr.io/crusoecloud/nccl-tests:13.0.1-ubuntu24.04-nccl-2.29.2-1 + imagePullPolicy: IfNotPresent + name: nccl-worker + securityContext: + capabilities: + add: ["IPC_LOCK"] + env: + - name: NCCL_TOPO_FILE + value: /opt/nccl_topo/b300-288gb-sxm-ib-cloud-hypervisor.xml + - name: UCX_RNDV_SCHEME + value: "get_zcopy" # UCX memory setting + - name: UCX_TLS + value: "self,sm,cuda_copy" # UCX memory setting + volumeMounts: + - mountPath: /dev/shm + name: dshm + - name: nccl-topo + mountPath: /opt/nccl_topo + resources: + limits: + nvidia.com/gpu: 8 # 8 GPUs per node + nvidia.com/hostdev: 8 + memory: 128000Mi + requests: + nvidia.com/gpu: 8 # 8 GPUs per node + nvidia.com/hostdev: 8 + memory: 128000Mi \ No newline at end of file From 6bbebfa1b949d831a2a9e02949103f02f64b40a7 Mon Sep 17 00:00:00 2001 From: Young Jeong Date: Wed, 16 Sep 2026 10:25:42 -0700 Subject: [PATCH 2/3] readme edit --- cmk-nccltests/README.md | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/cmk-nccltests/README.md b/cmk-nccltests/README.md index 6cb69c9..4e9bc1e 100644 --- a/cmk-nccltests/README.md +++ b/cmk-nccltests/README.md @@ -43,5 +43,6 @@ This directory contains Kubernetes manifests for running [NCCL tests](https://gi ## Configuration notes -- **Topology file**: B200, H100, and H200 jobs mount `/etc/crusoe/nccl_topo` from the host and set `NCCL_TOPO_FILE` to the SKU-specific XML. This path is pre-populated on Crusoe GPU nodes. -- **CUDA image**: B200 and H200 use `nccl-tests:13.0.1-ubuntu24.04-nccl-2.29.2-1`; H100 uses `nccl-tests:12.8.1-ubuntu24.04-nccl-2.26.5-1`. Both images are hosted at `ghcr.io/crusoecloud/nccl-tests`. \ No newline at end of file +- **Topology file**: B300, B200, H100, and H200 jobs mount `/etc/crusoe/nccl_topo` from the host and set `NCCL_TOPO_FILE` to the SKU-specific XML. This path is pre-populated on Crusoe GPU nodes. +- **CUDA image**: B200 and H200 use `nccl-tests:13.0.1-ubuntu24.04-nccl-2.29.2-1`; H100 uses `nccl-tests:12.8.1-ubuntu24.04-nccl-2.26.5-1`. Both images are hosted at `ghcr.io/crusoecloud/nccl-tests`. +- **B300 No Hyperthreading**: Certain B300 nodes provided by Crusoe disables Hyperthreading for certain CPU performance profile. Those VMs have a dedicated topology file, and therefore should use its own yaml file as indicated by `nccl-b300-noht.yaml`. \ No newline at end of file From 8be71cedf00d64dfa4c69a8215a36cb5fbdfd039 Mon Sep 17 00:00:00 2001 From: Young Jeong Date: Wed, 16 Sep 2026 10:28:51 -0700 Subject: [PATCH 3/3] small edit --- cmk-nccltests/README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cmk-nccltests/README.md b/cmk-nccltests/README.md index 3f1ca64..7b1436c 100644 --- a/cmk-nccltests/README.md +++ b/cmk-nccltests/README.md @@ -6,7 +6,7 @@ This directory contains Kubernetes manifests for running [NCCL tests](https://gi |---|---|---| | [nccl-b200.yaml](nccl-b200.yaml) | B200 180GB SXM | 8 | | [nccl-b300.yaml](nccl-b300.yaml) | B300 280GB SXM | 8 | -| [nccl-b300-noht.yaml](nccl-b300-nohtt.yaml) | B300 280GB SXM (No Hyperthreading) | 8 | +| [nccl-b300-noht.yaml](nccl-b300-noht.yaml) | B300 280GB SXM (No Hyperthreading) | 8 | | [nccl-gb200.yaml](nccl-gb200.yaml) | GB200 (NVL72) | 4 | | [nccl-h100.yaml](nccl-h100.yaml) | H100 80GB SXM | 8 | | [nccl-h200.yaml](nccl-h200.yaml) | H200 141GB SXM | 8 |