-
Notifications
You must be signed in to change notification settings - Fork 2
/
Copy pathdcgm-exporter.yaml
77 lines (77 loc) · 2.12 KB
/
dcgm-exporter.yaml
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
apiVersion: apps/v1
kind: DaemonSet
metadata:
name: "dcgm-exporter"
labels:
app.kubernetes.io/name: "dcgm-exporter"
app.kubernetes.io/version: "2.4.0-rc.2"
spec:
updateStrategy:
type: RollingUpdate
selector:
matchLabels:
app.kubernetes.io/name: "dcgm-exporter"
app.kubernetes.io/version: "2.4.0-rc.2"
template:
metadata:
labels:
app.kubernetes.io/name: "dcgm-exporter"
app.kubernetes.io/version: "2.4.0-rc.2"
name: "dcgm-exporter"
spec:
nodeSelector:
accelerator: nvidia-gpu
tolerations:
- key: "nvidia.com/gpu"
value: "present"
effect: "NoSchedule"
containers:
- image: "nvcr.io/nvidia/k8s/dcgm-exporter:2.1.8-2.4.0-rc.2-ubuntu18.04"
env:
- name: "DCGM_EXPORTER_LISTEN"
value: ":9400"
- name: "DCGM_EXPORTER_KUBERNETES"
value: "true"
- name: LD_LIBRARY_PATH
value: "/bin/nvidia/lib64/"
args: ["-c", "10000", "-f", "/etc/dcgm-exporter/1.x-compatibility-metrics.csv", "--kubernetes-gpu-id-type", "device-name"]
name: "dcgm-exporter"
ports:
- name: "metrics"
containerPort: 9400
securityContext:
capabilities:
add:
- SYS_ADMIN
privileged: true
runAsNonRoot: false
runAsUser: 0
volumeMounts:
- name: "pod-gpu-resources"
readOnly: true
mountPath: "/var/lib/kubelet/pod-resources"
- name: libnvidia
readOnly: true
mountPath: /bin/nvidia/lib64/
volumes:
- name: "pod-gpu-resources"
hostPath:
path: "/var/lib/kubelet/pod-resources"
- name: libnvidia
hostPath:
path: /home/kubernetes/bin/nvidia/lib64/ # for k3s: /usr/local/cuda-11.2/lib64/
---
kind: Service
apiVersion: v1
metadata:
name: "dcgm-exporter"
labels:
app.kubernetes.io/name: "dcgm-exporter"
app.kubernetes.io/version: "2.4.0-rc.2"
spec:
selector:
app.kubernetes.io/name: "dcgm-exporter"
app.kubernetes.io/version: "2.4.0-rc.2"
ports:
- name: "metrics"
port: 9400