-
Notifications
You must be signed in to change notification settings - Fork 166
/
autoscale_job.yaml
92 lines (92 loc) · 2.92 KB
/
autoscale_job.yaml
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
---
apiVersion: elastic.iml.github.io/v1alpha1
kind: ElasticJob
metadata:
name: deepctr-auto-scale
namespace: dlrover
spec:
distributionStrategy: ParameterServerStrategy
resourceLimits:
cpu: "15"
memory: "30000Mi"
replicaSpecs:
ps:
autoScale: True
template:
spec:
restartPolicy: Never
containers:
- name: main
# yamllint disable-line rule:line-length
image: registry.cn-hangzhou.aliyuncs.com/intell-ai/dlrover:deeprec_criteo_v2
imagePullPolicy: Always
command:
- /bin/bash
- -c
- "cd ./examples/tensorflow/criteo_deeprec \
&& python -m dlrover.trainer.entry.local_entry \
--platform=Kubernetes --conf=train_conf.TrainConf \
--enable_auto_scaling=True"
volumeMounts:
- name: pvc-nas
mountPath: /nas
volumes:
- name: pvc-nas
persistentVolumeClaim:
claimName: pvc-nas
worker:
autoScale: True
template:
spec:
restartPolicy: Never
containers:
- name: main
# yamllint disable-line rule:line-length
image: registry.cn-hangzhou.aliyuncs.com/intell-ai/dlrover:deeprec_criteo_v2
imagePullPolicy: Always
command:
- /bin/bash
- -c
- "cd ./examples/tensorflow/criteo_deeprec \
&& python -m dlrover.trainer.entry.local_entry \
--platform=Kubernetes --conf=train_conf.TrainConf \
--enable_auto_scaling=True"
volumeMounts:
- name: pvc-nas
mountPath: /nas
volumes:
- name: pvc-nas
persistentVolumeClaim:
claimName: pvc-nas
evaluator:
autoScale: True
replicas: 1
template:
spec:
restartPolicy: Never
containers:
- name: main
# yamllint disable-line rule:line-length
image: registry.cn-hangzhou.aliyuncs.com/intell-ai/dlrover:deeprec_criteo_v2
imagePullPolicy: Always
command:
- /bin/bash
- -c
- "cd ./examples/tensorflow/criteo_deeprec \
&& python -m dlrover.trainer.entry.local_entry \
--platform=Kubernetes --conf=train_conf.TrainConf \
--enable_auto_scaling=True"
volumeMounts:
- name: pvc-nas
mountPath: /nas
volumes:
- name: pvc-nas
persistentVolumeClaim:
claimName: pvc-nas
dlrover-master:
template:
spec:
restartPolicy: Never
containers:
- name: main
imagePullPolicy: Always