-
Notifications
You must be signed in to change notification settings - Fork 0
/
Copy pathtorch.yaml
101 lines (99 loc) · 2.98 KB
/
torch.yaml
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
apiVersion: "kubeflow.org/v1"
kind: PyTorchJob
metadata:
name: torch-dist
namespace: ml0
spec:
pytorchReplicaSpecs:
Master:
replicas: 1
restartPolicy: OnFailure
template:
metadata:
labels:
app: workerl
annotations:
sidecar.istio.io/inject: "false"
spec:
containers:
- name: pytorch
image: registry.i.sumus.work/kube/mlsynth:pytorch-2.0.1-cuda11.7-cudnn8-runtime
imagePullPolicy: Always
# command:
# - "python3"
# - "/opt/pytorch-mnist/mnist.py"
# - "--epochs=1"
command: ["/bin/bash", "-c"]
args: ["source a.sh"]
# args: ["sleep 6000"]
workingDir: /workspace
volumeMounts:
- name: datasets
mountPath: /datasets
- name: workspace
mountPath: /workspace
resources:
limits:
nvidia.com/gpu: 1
volumes:
- name: datasets
nfs:
server: k8s-st04.i.clive.tk
path: /exports
readOnly: true
- name: workspace
nfs:
server: kube-exp-w1.k8s.sumus.work
path: /mnt/ssd/workspace
readOnly: false
Worker:
replicas: 2
restartPolicy: OnFailure
template:
metadata:
labels:
app: workerl
annotations:
sidecar.istio.io/inject: "false"
spec:
containers:
- name: pytorch
image: registry.i.sumus.work/kube/mlsynth:pytorch-2.0.1-cuda11.7-cudnn8-runtime
imagePullPolicy: Always
# command:
# - "python3"
# - "/opt/pytorch-mnist/mnist.py"
# - "--epochs=1"
command: ["/bin/bash", "-c"]
args: ["source a.sh"]
# args: ["sleep 6000"]
workingDir: /workspace
volumeMounts:
- name: datasets
mountPath: /datasets
- name: workspace
mountPath: /workspace
resources:
limits:
nvidia.com/gpu: 1
affinity:
podAntiAffinity:
requiredDuringSchedulingIgnoredDuringExecution:
- labelSelector:
matchExpressions:
- key: app
operator: In
values:
- workerl
topologyKey: kubernetes.io/hostname
volumes:
- name: datasets
nfs:
server: k8s-st04.i.clive.tk
path: /exports
readOnly: true
- name: workspace
nfs:
server: kube-exp-w1.k8s.sumus.work
path: /mnt/ssd/workspace
readOnly: false