Fix Collection Service Helm charts package
[demo.git] / vnfs / DAaaS / prometheus-operator / templates / alertmanager / rules / kubernetes-system.yaml
1 # Generated from 'kubernetes-system' group from https://raw.githubusercontent.com/coreos/prometheus-operator/master/contrib/kube-prometheus/manifests/prometheus-rules.yaml
2 {{- if and .Values.defaultRules.create }}
3 apiVersion: {{ printf "%s/v1" (.Values.prometheusOperator.crdApiGroup | default "monitoring.coreos.com") }}
4 kind: PrometheusRule
5 metadata:
6   name: {{ printf "%s-%s" (include "prometheus-operator.fullname" .) "kubernetes-system" | trunc 63 | trimSuffix "-" }}
7   labels:
8     app: {{ template "prometheus-operator.name" . }}
9 {{ include "prometheus-operator.labels" . | indent 4 }}
10 {{- if .Values.defaultRules.labels }}
11 {{ toYaml .Values.defaultRules.labels | indent 4 }}
12 {{- end }}
13 {{- if .Values.defaultRules.annotations }}
14   annotations:
15 {{ toYaml .Values.defaultRules.annotations | indent 4 }}
16 {{- end }}
17 spec:
18   groups:
19   - name: kubernetes-system
20     rules:
21     - alert: KubeNodeNotReady
22       annotations:
23         message: '{{`{{ $labels.node }}`}} has been unready for more than an hour.'
24         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubenodenotready
25       expr: kube_node_status_condition{job="kube-state-metrics",condition="Ready",status="true"} == 0
26       for: 1h
27       labels:
28         severity: warning
29     - alert: KubeVersionMismatch
30       annotations:
31         message: There are {{`{{ $value }}`}} different versions of Kubernetes components running.
32         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeversionmismatch
33       expr: count(count(kubernetes_build_info{job!="kube-dns"}) by (gitVersion)) > 1
34       for: 1h
35       labels:
36         severity: warning
37     - alert: KubeClientErrors
38       annotations:
39         message: Kubernetes API server client '{{`{{ $labels.job }}`}}/{{`{{ $labels.instance }}`}}' is experiencing {{`{{ printf "%0.0f" $value }}`}}% errors.'
40         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeclienterrors
41       expr: |-
42         (sum(rate(rest_client_requests_total{code=~"5.."}[5m])) by (instance, job)
43           /
44         sum(rate(rest_client_requests_total[5m])) by (instance, job))
45         * 100 > 1
46       for: 15m
47       labels:
48         severity: warning
49     - alert: KubeClientErrors
50       annotations:
51         message: Kubernetes API server client '{{`{{ $labels.job }}`}}/{{`{{ $labels.instance }}`}}' is experiencing {{`{{ printf "%0.0f" $value }}`}} errors / second.
52         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeclienterrors
53       expr: sum(rate(ksm_scrape_error_total{job="kube-state-metrics"}[5m])) by (instance, job) > 0.1
54       for: 15m
55       labels:
56         severity: warning
57     - alert: KubeletTooManyPods
58       annotations:
59         message: Kubelet {{`{{ $labels.instance }}`}} is running {{`{{ $value }}`}} Pods, close to the limit of 110.
60         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubelettoomanypods
61       expr: kubelet_running_pod_count{job="kubelet"} > 110 * 0.9
62       for: 15m
63       labels:
64         severity: warning
65     - alert: KubeAPILatencyHigh
66       annotations:
67         message: The API server has a 99th percentile latency of {{`{{ $value }}`}} seconds for {{`{{ $labels.verb }}`}} {{`{{ $labels.resource }}`}}.
68         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeapilatencyhigh
69       expr: cluster_quantile:apiserver_request_latencies:histogram_quantile{job="apiserver",quantile="0.99",subresource!="log",verb!~"^(?:LIST|WATCH|WATCHLIST|PROXY|CONNECT)$"} > 1
70       for: 10m
71       labels:
72         severity: warning
73     - alert: KubeAPILatencyHigh
74       annotations:
75         message: The API server has a 99th percentile latency of {{`{{ $value }}`}} seconds for {{`{{ $labels.verb }}`}} {{`{{ $labels.resource }}`}}.
76         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeapilatencyhigh
77       expr: cluster_quantile:apiserver_request_latencies:histogram_quantile{job="apiserver",quantile="0.99",subresource!="log",verb!~"^(?:LIST|WATCH|WATCHLIST|PROXY|CONNECT)$"} > 4
78       for: 10m
79       labels:
80         severity: critical
81     - alert: KubeAPIErrorsHigh
82       annotations:
83         message: API server is returning errors for {{`{{ $value }}`}}% of requests.
84         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeapierrorshigh
85       expr: |-
86         sum(rate(apiserver_request_count{job="apiserver",code=~"^(?:5..)$"}[5m])) without(instance, pod)
87           /
88         sum(rate(apiserver_request_count{job="apiserver"}[5m])) without(instance, pod) * 100 > 10
89       for: 10m
90       labels:
91         severity: critical
92     - alert: KubeAPIErrorsHigh
93       annotations:
94         message: API server is returning errors for {{`{{ $value }}`}}% of requests.
95         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeapierrorshigh
96       expr: |-
97         sum(rate(apiserver_request_count{job="apiserver",code=~"^(?:5..)$"}[5m])) without(instance, pod)
98           /
99         sum(rate(apiserver_request_count{job="apiserver"}[5m])) without(instance, pod) * 100 > 5
100       for: 10m
101       labels:
102         severity: warning
103     - alert: KubeClientCertificateExpiration
104       annotations:
105         message: Kubernetes API certificate is expiring in less than 7 days.
106         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeclientcertificateexpiration
107       expr: histogram_quantile(0.01, sum by (job, le) (rate(apiserver_client_certificate_expiration_seconds_bucket{job="apiserver"}[5m]))) < 604800
108       labels:
109         severity: warning
110     - alert: KubeClientCertificateExpiration
111       annotations:
112         message: Kubernetes API certificate is expiring in less than 24 hours.
113         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeclientcertificateexpiration
114       expr: histogram_quantile(0.01, sum by (job, le) (rate(apiserver_client_certificate_expiration_seconds_bucket{job="apiserver"}[5m]))) < 86400
115       labels:
116         severity: critical
117 {{- end }}