Collectd operator utilties
[demo.git] / vnfs / DAaaS / operator / charts / prometheus-operator / templates / prometheus / rules / kubernetes-system.yaml
1 # Generated from 'kubernetes-system' group from https://raw.githubusercontent.com/coreos/prometheus-operator/master/contrib/kube-prometheus/manifests/prometheus-rules.yaml
2 # Do not change in-place! In order to change this file first read following link:
3 # https://github.com/helm/charts/tree/master/stable/prometheus-operator/hack
4 {{- if and .Values.defaultRules.create .Values.defaultRules.rules.kubernetesSystem }}
5 apiVersion: {{ printf "%s/v1" (.Values.prometheusOperator.crdApiGroup | default "monitoring.coreos.com") }}
6 kind: PrometheusRule
7 metadata:
8   name: {{ printf "%s-%s" (include "prometheus-operator.fullname" .) "kubernetes-system" | trunc 63 | trimSuffix "-" }}
9   labels:
10     app: {{ template "prometheus-operator.name" . }}
11 {{ include "prometheus-operator.labels" . | indent 4 }}
12 {{- if .Values.defaultRules.labels }}
13 {{ toYaml .Values.defaultRules.labels | indent 4 }}
14 {{- end }}
15 {{- if .Values.defaultRules.annotations }}
16   annotations:
17 {{ toYaml .Values.defaultRules.annotations | indent 4 }}
18 {{- end }}
19 spec:
20   groups:
21   - name: kubernetes-system
22     rules:
23     - alert: KubeNodeNotReady
24       annotations:
25         message: '{{`{{ $labels.node }}`}} has been unready for more than an hour.'
26         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubenodenotready
27       expr: kube_node_status_condition{job="kube-state-metrics",condition="Ready",status="true"} == 0
28       for: 1h
29       labels:
30         severity: warning
31     - alert: KubeVersionMismatch
32       annotations:
33         message: There are {{`{{ $value }}`}} different semantic versions of Kubernetes components running.
34         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeversionmismatch
35       expr: count(count by (gitVersion) (label_replace(kubernetes_build_info{job!="kube-dns"},"gitVersion","$1","gitVersion","(v[0-9]*.[0-9]*.[0-9]*).*"))) > 1
36       for: 1h
37       labels:
38         severity: warning
39     - alert: KubeClientErrors
40       annotations:
41         message: Kubernetes API server client '{{`{{ $labels.job }}`}}/{{`{{ $labels.instance }}`}}' is experiencing {{`{{ printf "%0.0f" $value }}`}}% errors.'
42         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeclienterrors
43       expr: |-
44         (sum(rate(rest_client_requests_total{code=~"5.."}[5m])) by (instance, job)
45           /
46         sum(rate(rest_client_requests_total[5m])) by (instance, job))
47         * 100 > 1
48       for: 15m
49       labels:
50         severity: warning
51     - alert: KubeClientErrors
52       annotations:
53         message: Kubernetes API server client '{{`{{ $labels.job }}`}}/{{`{{ $labels.instance }}`}}' is experiencing {{`{{ printf "%0.0f" $value }}`}} errors / second.
54         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeclienterrors
55       expr: sum(rate(ksm_scrape_error_total{job="kube-state-metrics"}[5m])) by (instance, job) > 0.1
56       for: 15m
57       labels:
58         severity: warning
59     - alert: KubeletTooManyPods
60       annotations:
61         message: Kubelet {{`{{ $labels.instance }}`}} is running {{`{{ $value }}`}} Pods, close to the limit of 110.
62         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubelettoomanypods
63       expr: kubelet_running_pod_count{job="kubelet"} > 110 * 0.9
64       for: 15m
65       labels:
66         severity: warning
67     - alert: KubeAPILatencyHigh
68       annotations:
69         message: The API server has a 99th percentile latency of {{`{{ $value }}`}} seconds for {{`{{ $labels.verb }}`}} {{`{{ $labels.resource }}`}}.
70         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeapilatencyhigh
71       expr: cluster_quantile:apiserver_request_latencies:histogram_quantile{job="apiserver",quantile="0.99",subresource!="log",verb!~"^(?:LIST|WATCH|WATCHLIST|PROXY|CONNECT)$"} > 1
72       for: 10m
73       labels:
74         severity: warning
75     - alert: KubeAPILatencyHigh
76       annotations:
77         message: The API server has a 99th percentile latency of {{`{{ $value }}`}} seconds for {{`{{ $labels.verb }}`}} {{`{{ $labels.resource }}`}}.
78         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeapilatencyhigh
79       expr: cluster_quantile:apiserver_request_latencies:histogram_quantile{job="apiserver",quantile="0.99",subresource!="log",verb!~"^(?:LIST|WATCH|WATCHLIST|PROXY|CONNECT)$"} > 4
80       for: 10m
81       labels:
82         severity: critical
83     - alert: KubeAPIErrorsHigh
84       annotations:
85         message: API server is returning errors for {{`{{ $value }}`}}% of requests.
86         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeapierrorshigh
87       expr: |-
88         sum(rate(apiserver_request_count{job="apiserver",code=~"^(?:5..)$"}[5m])) without(instance, pod)
89           /
90         sum(rate(apiserver_request_count{job="apiserver"}[5m])) without(instance, pod) * 100 > 10
91       for: 10m
92       labels:
93         severity: critical
94     - alert: KubeAPIErrorsHigh
95       annotations:
96         message: API server is returning errors for {{`{{ $value }}`}}% of requests.
97         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeapierrorshigh
98       expr: |-
99         sum(rate(apiserver_request_count{job="apiserver",code=~"^(?:5..)$"}[5m])) without(instance, pod)
100           /
101         sum(rate(apiserver_request_count{job="apiserver"}[5m])) without(instance, pod) * 100 > 5
102       for: 10m
103       labels:
104         severity: warning
105     - alert: KubeClientCertificateExpiration
106       annotations:
107         message: A client certificate used to authenticate to the apiserver is expiring in less than 7 days.
108         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeclientcertificateexpiration
109       expr: histogram_quantile(0.01, sum by (job, le) (rate(apiserver_client_certificate_expiration_seconds_bucket{job="apiserver"}[5m]))) < 604800
110       labels:
111         severity: warning
112     - alert: KubeClientCertificateExpiration
113       annotations:
114         message: A client certificate used to authenticate to the apiserver is expiring in less than 24 hours.
115         runbook_url: https://github.com/kubernetes-monitoring/kubernetes-mixin/tree/master/runbook.md#alert-name-kubeclientcertificateexpiration
116       expr: histogram_quantile(0.01, sum by (job, le) (rate(apiserver_client_certificate_expiration_seconds_bucket{job="apiserver"}[5m]))) < 86400
117       labels:
118         severity: critical
119 {{- end }}