Update integration config for InfiniBand GPU environment
- Add IB CNI annotation (hostdevice-net), /dev/shm, /dev/infiniband mounts - Add nvidia.com/hostdev: 2 resource for IB NICs - Add privileged securityContext with IPC_LOCK for NCCL - Add NCCL env vars (NCCL_DEBUG, NCCL_IB_DISABLE=0, NCCL_SOCKET_IFNAME) - Set image to nvcr.io/nvidia/pytorch:24.10-py3 - Set numNodes: 2 for multi-node distributed training Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
parent
6b72590351
commit
f62425843b
|
|
@ -12,10 +12,11 @@ spec:
|
||||||
weight: 1
|
weight: 1
|
||||||
reclaimable: true
|
reclaimable: true
|
||||||
capability:
|
capability:
|
||||||
# TODO: Adjust based on your cluster capacity
|
# 3노드 x GPU 8장 = 24, hostdev 2개 = 6
|
||||||
cpu: "100"
|
cpu: "100"
|
||||||
memory: "512Gi"
|
memory: "512Gi"
|
||||||
nvidia.com/gpu: "24"
|
nvidia.com/gpu: "24"
|
||||||
|
nvidia.com/hostdev: "6"
|
||||||
|
|
||||||
---
|
---
|
||||||
# 2. ClusterTrainingRuntime with Volcano gang scheduling + topology-aware scheduling
|
# 2. ClusterTrainingRuntime with Volcano gang scheduling + topology-aware scheduling
|
||||||
|
|
@ -30,7 +31,7 @@ spec:
|
||||||
mlPolicy:
|
mlPolicy:
|
||||||
torch:
|
torch:
|
||||||
numProcPerNode: 8
|
numProcPerNode: 8
|
||||||
numNodes: 1
|
numNodes: 2
|
||||||
# Volcano gang scheduling 활성화 - PodGroup 자동 생성
|
# Volcano gang scheduling 활성화 - PodGroup 자동 생성
|
||||||
podGroupPolicy:
|
podGroupPolicy:
|
||||||
volcano:
|
volcano:
|
||||||
|
|
@ -48,15 +49,47 @@ spec:
|
||||||
template:
|
template:
|
||||||
spec:
|
spec:
|
||||||
template:
|
template:
|
||||||
|
metadata:
|
||||||
|
annotations:
|
||||||
|
# InfiniBand CNI
|
||||||
|
k8s.v1.cni.cncf.io/networks: hostdevice-net
|
||||||
spec:
|
spec:
|
||||||
containers:
|
containers:
|
||||||
- name: node
|
- name: node
|
||||||
image: ghcr.io/kubeflow/trainer/torch-runtime:latest
|
image: nvcr.io/nvidia/pytorch:24.10-py3
|
||||||
|
securityContext:
|
||||||
|
privileged: true
|
||||||
|
capabilities:
|
||||||
|
add:
|
||||||
|
- IPC_LOCK
|
||||||
|
env:
|
||||||
|
- name: NCCL_DEBUG
|
||||||
|
value: "INFO"
|
||||||
|
- name: NCCL_IB_DISABLE
|
||||||
|
value: "0"
|
||||||
|
- name: NCCL_SOCKET_IFNAME
|
||||||
|
value: "net"
|
||||||
resources:
|
resources:
|
||||||
requests:
|
requests:
|
||||||
nvidia.com/gpu: "8"
|
nvidia.com/gpu: "8"
|
||||||
|
nvidia.com/hostdev: "2"
|
||||||
limits:
|
limits:
|
||||||
nvidia.com/gpu: "8"
|
nvidia.com/gpu: "8"
|
||||||
|
nvidia.com/hostdev: "2"
|
||||||
|
volumeMounts:
|
||||||
|
- name: shared-memory
|
||||||
|
mountPath: /dev/shm
|
||||||
|
- name: infiniband
|
||||||
|
mountPath: /dev/infiniband
|
||||||
|
volumes:
|
||||||
|
- name: shared-memory
|
||||||
|
emptyDir:
|
||||||
|
medium: Memory
|
||||||
|
sizeLimit: 128Gi
|
||||||
|
- name: infiniband
|
||||||
|
hostPath:
|
||||||
|
path: /dev/infiniband
|
||||||
|
type: Directory
|
||||||
|
|
||||||
---
|
---
|
||||||
# 3. Example TrainJob (설치 테스트용 - 단일노드)
|
# 3. Example TrainJob (설치 테스트용 - 단일노드)
|
||||||
|
|
@ -72,11 +105,12 @@ spec:
|
||||||
name: torch-distributed-volcano
|
name: torch-distributed-volcano
|
||||||
trainer:
|
trainer:
|
||||||
# TODO: Replace with your actual training image
|
# TODO: Replace with your actual training image
|
||||||
image: docker.io/your-org/your-training-image:latest
|
image: nvcr.io/nvidia/pytorch:24.10-py3
|
||||||
numNodes: 1
|
numNodes: 2
|
||||||
numProcPerNode: "8"
|
|
||||||
resourcesPerNode:
|
resourcesPerNode:
|
||||||
requests:
|
requests:
|
||||||
nvidia.com/gpu: "8"
|
nvidia.com/gpu: "8"
|
||||||
|
nvidia.com/hostdev: "2"
|
||||||
limits:
|
limits:
|
||||||
nvidia.com/gpu: "8"
|
nvidia.com/gpu: "8"
|
||||||
|
nvidia.com/hostdev: "2"
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue