diff --git a/class/defaults.yml b/class/defaults.yml index a34bbf2..b753e87 100644 --- a/class/defaults.yml +++ b/class/defaults.yml @@ -143,6 +143,10 @@ parameters: enabled: false config: {} overrides: {} + thanosRuler: + enabled: false + config: {} + overrides: {} prometheusOperator: enabled: true diff --git a/component/common.libsonnet b/component/common.libsonnet index c9ff792..b1f833b 100644 --- a/component/common.libsonnet +++ b/component/common.libsonnet @@ -41,6 +41,7 @@ local instanceComponents = [ 'prometheus', 'prometheusAdapter', 'kubePrometheus', + 'thanosRuler', ]; local imageIsDockerIOShort = function(image) @@ -233,6 +234,135 @@ local grafanaIngress(instanceName, instanceParams) = if instanceParams.grafana.i }, } else {}; +// NOTE(mdl): kube-prometheus ships no thanosRuler component +local thanosRuler(instanceName, instanceParams) = + local name = formatComponentName('thanosRuler', instanceName); + local namespace = instanceParams.common.namespace; + local metadata = { + name: name, + namespace: namespace, + }; + local config = instanceParams.thanosRuler.config; + + local serviceName(component) = + com.getValueOrDefault( + instanceParams[component].config, 'name', formatComponentName(component, instanceName) + ); + + local configuresEndpoint(fields) = std.any([ std.objectHas(config, f) for f in fields ]); + local endpointDefaults = + ( + if instanceParams.prometheus.enabled && !configuresEndpoint([ 'queryEndpoints', 'queryConfig' ]) then { + // NOTE: The CRD recommends using `queryConfig` for Thanos >= 0.11.0. + // Eventually the default config should probably detect the Thanos + // version and adjust the config accordingly. + queryEndpoints: [ 'http://prometheus-%s.%s.svc:9090' % [ serviceName('prometheus'), namespace ] ], + } else {} + ) + ( + if instanceParams.alertmanager.enabled && !configuresEndpoint([ 'alertmanagersUrl', 'alertmanagersConfig' ]) then { + // NOTE: The CRD recommends using `alertmanagersConfig` for Thanos >= + // 0.10.0. Eventually the default config should probably detect the + // Thanos version and adjust the config accordingly. + alertmanagersUrl: [ 'http://alertmanager-%s.%s.svc:9093' % [ serviceName('alertmanager'), namespace ] ], + } else {} + ); + + local spec = endpointDefaults + config { serviceAccountName: name }; + { + thanosRuler: if instanceParams.thanosRuler.enabled then + assert std.objectHas(spec, 'queryEndpoints') || std.objectHas(spec, 'queryConfig') : + 'thanosRuler of instance `%s` needs `config.queryEndpoints` or `config.queryConfig` when the instance has no Prometheus enabled' % instanceName; + { + serviceAccount: { + apiVersion: 'v1', + kind: 'ServiceAccount', + metadata: metadata, + }, + thanosRuler: { + apiVersion: 'monitoring.coreos.com/v1', + kind: 'ThanosRuler', + metadata: metadata, + spec: spec, + }, + serviceMonitor: { + apiVersion: 'monitoring.coreos.com/v1', + kind: 'ServiceMonitor', + metadata: metadata, + spec: { + namespaceSelector: { + matchNames: [ + namespace, + ], + }, + selector: { + matchLabels: { + 'operated-thanos-ruler': 'true', + }, + }, + endpoints: [ + { port: 'web' }, + ], + }, + }, + prometheusRule: { + apiVersion: 'monitoring.coreos.com/v1', + kind: 'PrometheusRule', + metadata: metadata, + spec: { + groups: [ { + name: name, + rules: [ + { + alert: 'ThanosRulerDown', + expr: 'up{namespace="%s",service="thanos-ruler-operated"} == 0 or absent(up{namespace="%s",service="thanos-ruler-operated"})' % [ namespace, namespace ], + 'for': '15m', + labels: { + severity: 'critical', + }, + annotations: { + summary: 'Thanos Ruler %s is down, its alert rules are not being evaluated.' % name, + }, + }, + { + alert: 'ThanosRulerRuleEvaluationFailing', + expr: 'increase(prometheus_rule_evaluation_failures_total{namespace="%s",service="thanos-ruler-operated"}[10m]) > 0' % namespace, + 'for': '15m', + labels: { + severity: 'warning', + }, + annotations: { + summary: 'Thanos Ruler %s is failing rule evaluations, alerts may be missed.' % name, + }, + }, + { + alert: 'ThanosRulerIsDroppingAlerts', + expr: 'sum(rate(thanos_alert_sender_alerts_dropped_total{namespace="%s"}[5m])) > 0 or sum(rate(thanos_alert_queue_alerts_dropped_total{namespace="%s"}[5m])) > 0' % [ namespace, namespace ], + 'for': '5m', + labels: { + severity: 'critical', + }, + annotations: { + summary: 'Thanos Ruler %s is dropping alerts instead of delivering them to Alertmanager.' % name, + }, + }, + { + alert: 'ThanosRulerNoEvaluation', + expr: 'time() - max by (rule_group) (prometheus_rule_group_last_evaluation_timestamp_seconds{namespace="%s",service="thanos-ruler-operated"}) > 10 * max by (rule_group) (prometheus_rule_group_interval_seconds{namespace="%s",service="thanos-ruler-operated"})' % [ namespace, namespace ], + 'for': '5m', + labels: { + severity: 'critical', + }, + annotations: { + summary: 'Thanos Ruler %s has not evaluated a rule group for 10 intervals.' % name, + }, + }, + ], + } ], + }, + }, + } else {}, + }; + local grafanaStorage(instanceName, instanceParams) = if instanceParams.grafana.persistence.enabled then assert instanceParams.grafana.persistence.size != '' : 'Storage size cannot be empty when persistence enabled'; { @@ -356,6 +486,7 @@ local stackForInstance = function(instanceName) }, } + patchGrafanaDataSource(instanceName) + patchKubeControlPlaneSelectors(instanceName) + com.makeMergeable(cm), } + + thanosRuler(instanceName, confWithBase) + grafanaStorage(instanceName, confWithBase) + grafanaIngress(instanceName, confWithBase) + addNodeExporterContainerArgs(instanceName, confWithBase) diff --git a/component/main.jsonnet b/component/main.jsonnet index 8bd3dd5..4970d2d 100644 --- a/component/main.jsonnet +++ b/component/main.jsonnet @@ -87,6 +87,7 @@ local secrets = std.foldl( local renderInstance = function(instanceName, stack) local prometheus = common.render_component(stack, 'prometheus', 20, instanceName); local alertmanager = common.render_component(stack, 'alertmanager', 30, instanceName); + local thanosRuler = common.render_component(stack, 'thanosRuler', 35, instanceName); local grafana = common.render_component(stack, 'grafana', 40, instanceName); local nodeExporter = common.render_component(stack, 'nodeExporter', 50, instanceName); local blackboxExporter = common.render_component(stack, 'blackboxExporter', 60, instanceName); @@ -104,7 +105,8 @@ local renderInstance = function(instanceName, stack) (if p.kubernetesControlPlane.enabled then kubernetesControlPlane else {}) + (if p.prometheusAdapter.enabled then prometheusAdapter else {}) + (if p.kubeStateMetrics.enabled then kubeStateMetrics else {}) + - (if p.kubePrometheus.enabled then kubePrometheus else {}) + (if p.kubePrometheus.enabled then kubePrometheus else {}) + + (if p.thanosRuler.enabled then thanosRuler else {}) ; diff --git a/docs/modules/ROOT/pages/references/parameters.adoc b/docs/modules/ROOT/pages/references/parameters.adoc index 0081e5b..644cff7 100644 --- a/docs/modules/ROOT/pages/references/parameters.adoc +++ b/docs/modules/ROOT/pages/references/parameters.adoc @@ -16,6 +16,7 @@ Multiple instances of the following components can be configured: * kubernetesControlPlane * prometheusAdapter * kubeStateMetrics +* thanosRuler == `kubernetes_version` @@ -255,6 +256,10 @@ kubePrometheus: enabled: false config: {} overrides: {} +thanosRuler: + enabled: false + config: {} + overrides: {} ---- The base configuration shared by all instances. @@ -381,6 +386,36 @@ This means configuration side effects don't apply and the configuration can cont The easiest way to find the allowed parameters is to look at the local `defaults` variable. See the kube state metrics defaults as an example: https://github.com/prometheus-operator/kube-prometheus/blob/aeb50f066eadf9831c53cdf9228e09dd4e9d28b2/jsonnet/kube-prometheus/components/kube-state-metrics.libsonnet#L7-L48[kube-prometheus/components/kube-state-metrics.libsonnet] +== `base.thanosRuler`, `instances.*.thanosRuler` + +[horizontal] +type:: dict +example:: ++ +[source,yaml] +---- +thanosRuler: + enabled: true + config: + replicas: 2 + queryEndpoints: + - http://prometheus-infra.syn-infra-monitoring:9090 + ruleNamespaceSelector: + matchExpressions: + - key: appcat.vshn.io/servicename + operator: Exists +---- + +`config` is merged into the https://prometheus-operator.dev/docs/api-reference/api/#monitoring.coreos.com/v1.ThanosRulerSpec[`ThanosRuler` spec]. + +If the same instance has `prometheus.enabled` or `alertmanager.enabled`, `spec.queryEndpoints` and `spec.alertmanagersUrl` default to the respective services of that instance. +An explicit value in `config` always wins, and the default is skipped entirely when `config` sets the mutually exclusive `queryConfig` resp. `alertmanagersConfig`. +Set them explicitly to evaluate against, or alert through, another instance or namespace. + +NOTE: Depending on the target cluster, evaluating against or alerting through an instance in another namespace may require custom network policies. + +Rendering fails if the instance has no Prometheus enabled and `config` sets neither `queryEndpoints` nor `queryConfig`. + == `instances.*.(prometheus|alertmanager|grafana).networkPolicy.additionalIngressRules` [horizontal] diff --git a/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_alertmanager.yaml b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_alertmanager.yaml new file mode 100644 index 0000000..f55f6ad --- /dev/null +++ b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_alertmanager.yaml @@ -0,0 +1,38 @@ +apiVersion: monitoring.coreos.com/v1 +kind: Alertmanager +metadata: + annotations: + source: https://github.com/projectsyn/component-prometheus + labels: + alertmanager: alertmanager-test + app.kubernetes.io/component: alert-router + app.kubernetes.io/managed-by: commodore + app.kubernetes.io/name: alertmanager + app.kubernetes.io/part-of: kube-prometheus + app.kubernetes.io/version: 0.22.2 + name: alertmanager-test + namespace: syn-prometheus +spec: + image: quay.io/prometheus/alertmanager:v0.22.2 + nodeSelector: + kubernetes.io/os: linux + podMetadata: + labels: + app.kubernetes.io/component: alert-router + app.kubernetes.io/name: alertmanager + app.kubernetes.io/part-of: kube-prometheus + app.kubernetes.io/version: 0.22.2 + replicas: 3 + resources: + limits: + cpu: 100m + memory: 100Mi + requests: + cpu: 4m + memory: 100Mi + securityContext: + fsGroup: 2000 + runAsNonRoot: true + runAsUser: 1000 + serviceAccountName: alertmanager-alertmanager-test + version: 0.22.2 diff --git a/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_podDisruptionBudget.yaml b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_podDisruptionBudget.yaml new file mode 100644 index 0000000..0e53f65 --- /dev/null +++ b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_podDisruptionBudget.yaml @@ -0,0 +1,21 @@ +apiVersion: policy/v1beta1 +kind: PodDisruptionBudget +metadata: + annotations: + source: https://github.com/projectsyn/component-prometheus + labels: + app.kubernetes.io/component: alert-router + app.kubernetes.io/managed-by: commodore + app.kubernetes.io/name: alertmanager + app.kubernetes.io/part-of: kube-prometheus + app.kubernetes.io/version: 0.22.2 + name: alertmanager-alertmanager-test + namespace: syn-prometheus +spec: + maxUnavailable: 1 + selector: + matchLabels: + alertmanager: alertmanager-test + app.kubernetes.io/component: alert-router + app.kubernetes.io/name: alertmanager + app.kubernetes.io/part-of: kube-prometheus diff --git a/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_prometheusRule.yaml b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_prometheusRule.yaml new file mode 100644 index 0000000..f911e3e --- /dev/null +++ b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_prometheusRule.yaml @@ -0,0 +1,177 @@ +apiVersion: monitoring.coreos.com/v1 +kind: PrometheusRule +metadata: + annotations: + source: https://github.com/projectsyn/component-prometheus + labels: + app.kubernetes.io/component: alert-router + app.kubernetes.io/managed-by: commodore + app.kubernetes.io/name: alertmanager + app.kubernetes.io/part-of: kube-prometheus + app.kubernetes.io/version: 0.22.2 + monitoring.syn.tools/enabled: 'true' + prometheus: test + role: alert-rules + name: alertmanager-alertmanager-test-rules + namespace: syn-prometheus +spec: + groups: + - name: alertmanager.rules + rules: + - alert: AlertmanagerFailedReload + annotations: + description: Configuration has failed to load for {{ $labels.namespace + }}/{{ $labels.pod}}. + runbook_url: https://runbooks.prometheus-operator.dev/runbooks/alertmanager/alertmanagerfailedreload + summary: Reloading an Alertmanager configuration has failed. + expr: | + # Without max_over_time, failed scrapes could create false negatives, see + # https://www.robustperception.io/alerting-on-gauges-in-prometheus-2-0 for details. + max_over_time(alertmanager_config_last_reload_successful{job="alertmanager-alertmanager-test",namespace="syn-prometheus"}[5m]) == 0 + for: 10m + labels: + severity: critical + syn: 'true' + syn_component: prometheus + - alert: AlertmanagerMembersInconsistent + annotations: + description: Alertmanager {{ $labels.namespace }}/{{ $labels.pod}} has + only found {{ $value }} members of the {{$labels.job}} cluster. + runbook_url: https://runbooks.prometheus-operator.dev/runbooks/alertmanager/alertmanagermembersinconsistent + summary: A member of an Alertmanager cluster has not found all other cluster + members. + expr: | + # Without max_over_time, failed scrapes could create false negatives, see + # https://www.robustperception.io/alerting-on-gauges-in-prometheus-2-0 for details. + max_over_time(alertmanager_cluster_members{job="alertmanager-alertmanager-test",namespace="syn-prometheus"}[5m]) + < on (namespace,service) group_left + count by (namespace,service) (max_over_time(alertmanager_cluster_members{job="alertmanager-alertmanager-test",namespace="syn-prometheus"}[5m])) + for: 15m + labels: + severity: critical + syn: 'true' + syn_component: prometheus + - alert: AlertmanagerFailedToSendAlerts + annotations: + description: Alertmanager {{ $labels.namespace }}/{{ $labels.pod}} failed + to send {{ $value | humanizePercentage }} of notifications to {{ $labels.integration + }}. + runbook_url: https://runbooks.prometheus-operator.dev/runbooks/alertmanager/alertmanagerfailedtosendalerts + summary: An Alertmanager instance failed to send notifications. + expr: | + ( + rate(alertmanager_notifications_failed_total{job="alertmanager-alertmanager-test",namespace="syn-prometheus"}[5m]) + / + rate(alertmanager_notifications_total{job="alertmanager-alertmanager-test",namespace="syn-prometheus"}[5m]) + ) + > 0.01 + for: 5m + labels: + severity: warning + syn: 'true' + syn_component: prometheus + - alert: AlertmanagerClusterFailedToSendAlerts + annotations: + description: The minimum notification failure rate to {{ $labels.integration + }} sent from any instance in the {{$labels.job}} cluster is {{ $value + | humanizePercentage }}. + runbook_url: https://runbooks.prometheus-operator.dev/runbooks/alertmanager/alertmanagerclusterfailedtosendalerts + summary: All Alertmanager instances in a cluster failed to send notifications + to a critical integration. + expr: | + min by (namespace,service, integration) ( + rate(alertmanager_notifications_failed_total{job="alertmanager-alertmanager-test",namespace="syn-prometheus", integration=~`.*`}[5m]) + / + rate(alertmanager_notifications_total{job="alertmanager-alertmanager-test",namespace="syn-prometheus", integration=~`.*`}[5m]) + ) + > 0.01 + for: 5m + labels: + severity: critical + syn: 'true' + syn_component: prometheus + - alert: AlertmanagerClusterFailedToSendAlerts + annotations: + description: The minimum notification failure rate to {{ $labels.integration + }} sent from any instance in the {{$labels.job}} cluster is {{ $value + | humanizePercentage }}. + runbook_url: https://runbooks.prometheus-operator.dev/runbooks/alertmanager/alertmanagerclusterfailedtosendalerts + summary: All Alertmanager instances in a cluster failed to send notifications + to a non-critical integration. + expr: | + min by (namespace,service, integration) ( + rate(alertmanager_notifications_failed_total{job="alertmanager-alertmanager-test",namespace="syn-prometheus", integration!~`.*`}[5m]) + / + rate(alertmanager_notifications_total{job="alertmanager-alertmanager-test",namespace="syn-prometheus", integration!~`.*`}[5m]) + ) + > 0.01 + for: 5m + labels: + severity: warning + syn: 'true' + syn_component: prometheus + - alert: AlertmanagerConfigInconsistent + annotations: + description: Alertmanager instances within the {{$labels.job}} cluster + have different configurations. + runbook_url: https://runbooks.prometheus-operator.dev/runbooks/alertmanager/alertmanagerconfiginconsistent + summary: Alertmanager instances within the same cluster have different + configurations. + expr: | + count by (namespace,service) ( + count_values by (namespace,service) ("config_hash", alertmanager_config_hash{job="alertmanager-alertmanager-test",namespace="syn-prometheus"}) + ) + != 1 + for: 20m + labels: + severity: critical + syn: 'true' + syn_component: prometheus + - alert: AlertmanagerClusterDown + annotations: + description: '{{ $value | humanizePercentage }} of Alertmanager instances + within the {{$labels.job}} cluster have been up for less than half of + the last 5m.' + runbook_url: https://runbooks.prometheus-operator.dev/runbooks/alertmanager/alertmanagerclusterdown + summary: Half or more of the Alertmanager instances within the same cluster + are down. + expr: | + ( + count by (namespace,service) ( + avg_over_time(up{job="alertmanager-alertmanager-test",namespace="syn-prometheus"}[5m]) < 0.5 + ) + / + count by (namespace,service) ( + up{job="alertmanager-alertmanager-test",namespace="syn-prometheus"} + ) + ) + >= 0.5 + for: 5m + labels: + severity: critical + syn: 'true' + syn_component: prometheus + - alert: AlertmanagerClusterCrashlooping + annotations: + description: '{{ $value | humanizePercentage }} of Alertmanager instances + within the {{$labels.job}} cluster have restarted at least 5 times in + the last 10m.' + runbook_url: https://runbooks.prometheus-operator.dev/runbooks/alertmanager/alertmanagerclustercrashlooping + summary: Half or more of the Alertmanager instances within the same cluster + are crashlooping. + expr: | + ( + count by (namespace,service) ( + changes(process_start_time_seconds{job="alertmanager-alertmanager-test",namespace="syn-prometheus"}[10m]) > 4 + ) + / + count by (namespace,service) ( + up{job="alertmanager-alertmanager-test",namespace="syn-prometheus"} + ) + ) + >= 0.5 + for: 5m + labels: + severity: critical + syn: 'true' + syn_component: prometheus diff --git a/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_secret.yaml b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_secret.yaml new file mode 100644 index 0000000..a70b7b9 --- /dev/null +++ b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_secret.yaml @@ -0,0 +1,29 @@ +apiVersion: v1 +kind: Secret +metadata: + annotations: + source: https://github.com/projectsyn/component-prometheus + labels: + alertmanager: alertmanager-test + app.kubernetes.io/component: alert-router + app.kubernetes.io/managed-by: commodore + app.kubernetes.io/name: alertmanager + app.kubernetes.io/part-of: kube-prometheus + app.kubernetes.io/version: 0.22.2 + name: alertmanager-alertmanager-test + namespace: syn-prometheus +stringData: + alertmanager.yaml: |- + "global": + "resolve_timeout": "5m" + "inhibit_rules": [] + "receivers": [] + "route": + "group_by": + - "namespace" + "group_interval": "5m" + "group_wait": "30s" + "receiver": "Default" + "repeat_interval": "12h" + "routes": [] +type: Opaque diff --git a/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_service.yaml b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_service.yaml new file mode 100644 index 0000000..fd32fd8 --- /dev/null +++ b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_service.yaml @@ -0,0 +1,26 @@ +apiVersion: v1 +kind: Service +metadata: + annotations: + source: https://github.com/projectsyn/component-prometheus + labels: + alertmanager: alertmanager-test + app.kubernetes.io/component: alert-router + app.kubernetes.io/managed-by: commodore + app.kubernetes.io/name: alertmanager + app.kubernetes.io/part-of: kube-prometheus + app.kubernetes.io/version: 0.22.2 + name: alertmanager-alertmanager-test + namespace: syn-prometheus +spec: + ports: + - name: web + port: 9093 + targetPort: web + selector: + alertmanager: alertmanager-test + app: alertmanager + app.kubernetes.io/component: alert-router + app.kubernetes.io/name: alertmanager + app.kubernetes.io/part-of: kube-prometheus + sessionAffinity: ClientIP diff --git a/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_serviceAccount.yaml b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_serviceAccount.yaml new file mode 100644 index 0000000..451e83d --- /dev/null +++ b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_serviceAccount.yaml @@ -0,0 +1,14 @@ +apiVersion: v1 +kind: ServiceAccount +metadata: + annotations: + source: https://github.com/projectsyn/component-prometheus + labels: + alertmanager: alertmanager-test + app.kubernetes.io/component: alert-router + app.kubernetes.io/managed-by: commodore + app.kubernetes.io/name: alertmanager + app.kubernetes.io/part-of: kube-prometheus + app.kubernetes.io/version: 0.22.2 + name: alertmanager-alertmanager-test + namespace: syn-prometheus diff --git a/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_serviceMonitor.yaml b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_serviceMonitor.yaml new file mode 100644 index 0000000..b7298b1 --- /dev/null +++ b/tests/golden/thanos/prometheus/prometheus/30_test_alertmanager_serviceMonitor.yaml @@ -0,0 +1,23 @@ +apiVersion: monitoring.coreos.com/v1 +kind: ServiceMonitor +metadata: + annotations: + source: https://github.com/projectsyn/component-prometheus + labels: + app.kubernetes.io/component: alert-router + app.kubernetes.io/managed-by: commodore + app.kubernetes.io/name: alertmanager + app.kubernetes.io/part-of: kube-prometheus + app.kubernetes.io/version: 0.22.2 + name: alertmanager + namespace: syn-prometheus +spec: + endpoints: + - interval: 30s + port: web + selector: + matchLabels: + alertmanager: alertmanager-test + app.kubernetes.io/component: alert-router + app.kubernetes.io/name: alertmanager + app.kubernetes.io/part-of: kube-prometheus diff --git a/tests/golden/thanos/prometheus/prometheus/35_test_thanosRuler_prometheusRule.yaml b/tests/golden/thanos/prometheus/prometheus/35_test_thanosRuler_prometheusRule.yaml new file mode 100644 index 0000000..c0cce0c --- /dev/null +++ b/tests/golden/thanos/prometheus/prometheus/35_test_thanosRuler_prometheusRule.yaml @@ -0,0 +1,52 @@ +apiVersion: monitoring.coreos.com/v1 +kind: PrometheusRule +metadata: + annotations: + source: https://github.com/projectsyn/component-prometheus + labels: + app.kubernetes.io/managed-by: commodore + app.kubernetes.io/part-of: syn + monitoring.syn.tools/enabled: 'true' + name: thanosruler-test + namespace: syn-prometheus +spec: + groups: + - name: thanosruler-test + rules: + - alert: ThanosRulerDown + annotations: + summary: Thanos Ruler thanosruler-test is down, its alert rules are not + being evaluated. + expr: up{namespace="syn-prometheus",service="thanos-ruler-operated"} == + 0 or absent(up{namespace="syn-prometheus",service="thanos-ruler-operated"}) + for: 15m + labels: + severity: critical + - alert: ThanosRulerRuleEvaluationFailing + annotations: + summary: Thanos Ruler thanosruler-test is failing rule evaluations, alerts + may be missed. + expr: increase(prometheus_rule_evaluation_failures_total{namespace="syn-prometheus",service="thanos-ruler-operated"}[10m]) + > 0 + for: 15m + labels: + severity: warning + - alert: ThanosRulerIsDroppingAlerts + annotations: + summary: Thanos Ruler thanosruler-test is dropping alerts instead of delivering + them to Alertmanager. + expr: sum(rate(thanos_alert_sender_alerts_dropped_total{namespace="syn-prometheus"}[5m])) + > 0 or sum(rate(thanos_alert_queue_alerts_dropped_total{namespace="syn-prometheus"}[5m])) + > 0 + for: 5m + labels: + severity: critical + - alert: ThanosRulerNoEvaluation + annotations: + summary: Thanos Ruler thanosruler-test has not evaluated a rule group + for 10 intervals. + expr: time() - max by (rule_group) (prometheus_rule_group_last_evaluation_timestamp_seconds{namespace="syn-prometheus",service="thanos-ruler-operated"}) + > 10 * max by (rule_group) (prometheus_rule_group_interval_seconds{namespace="syn-prometheus",service="thanos-ruler-operated"}) + for: 5m + labels: + severity: critical diff --git a/tests/golden/thanos/prometheus/prometheus/35_test_thanosRuler_serviceAccount.yaml b/tests/golden/thanos/prometheus/prometheus/35_test_thanosRuler_serviceAccount.yaml new file mode 100644 index 0000000..025cedc --- /dev/null +++ b/tests/golden/thanos/prometheus/prometheus/35_test_thanosRuler_serviceAccount.yaml @@ -0,0 +1,10 @@ +apiVersion: v1 +kind: ServiceAccount +metadata: + annotations: + source: https://github.com/projectsyn/component-prometheus + labels: + app.kubernetes.io/managed-by: commodore + app.kubernetes.io/part-of: syn + name: thanosruler-test + namespace: syn-prometheus diff --git a/tests/golden/thanos/prometheus/prometheus/35_test_thanosRuler_serviceMonitor.yaml b/tests/golden/thanos/prometheus/prometheus/35_test_thanosRuler_serviceMonitor.yaml new file mode 100644 index 0000000..bf97a99 --- /dev/null +++ b/tests/golden/thanos/prometheus/prometheus/35_test_thanosRuler_serviceMonitor.yaml @@ -0,0 +1,19 @@ +apiVersion: monitoring.coreos.com/v1 +kind: ServiceMonitor +metadata: + annotations: + source: https://github.com/projectsyn/component-prometheus + labels: + app.kubernetes.io/managed-by: commodore + app.kubernetes.io/part-of: syn + name: thanosruler-test + namespace: syn-prometheus +spec: + endpoints: + - port: web + namespaceSelector: + matchNames: + - syn-prometheus + selector: + matchLabels: + operated-thanos-ruler: 'true' diff --git a/tests/golden/thanos/prometheus/prometheus/35_test_thanosRuler_thanosRuler.yaml b/tests/golden/thanos/prometheus/prometheus/35_test_thanosRuler_thanosRuler.yaml new file mode 100644 index 0000000..d7fc24c --- /dev/null +++ b/tests/golden/thanos/prometheus/prometheus/35_test_thanosRuler_thanosRuler.yaml @@ -0,0 +1,22 @@ +apiVersion: monitoring.coreos.com/v1 +kind: ThanosRuler +metadata: + annotations: + source: https://github.com/projectsyn/component-prometheus + labels: + app.kubernetes.io/managed-by: commodore + app.kubernetes.io/part-of: syn + name: thanosruler-test + namespace: syn-prometheus +spec: + alertmanagersUrl: + - http://alertmanager-alertmanager-other.syn-other-monitoring:9093 + evaluationInterval: 1m + queryEndpoints: + - http://prometheus-test.syn-prometheus.svc:9090 + replicas: 1 + ruleNamespaceSelector: + matchExpressions: + - key: monitoring.example.com/user + operator: Exists + serviceAccountName: thanosruler-test diff --git a/tests/thanos.yml b/tests/thanos.yml index 6f2b062..06ba73f 100644 --- a/tests/thanos.yml +++ b/tests/thanos.yml @@ -38,3 +38,19 @@ parameters: objectStorageConfig: key: thanos name: thanos-objstore-config + alertmanager: + enabled: true + thanosRuler: + enabled: true + config: + replicas: 1 + alertmanagersUrl: + - http://alertmanager-alertmanager-other.syn-other-monitoring:9093 + ruleNamespaceSelector: + matchExpressions: + - key: monitoring.example.com/user + operator: Exists + overrides: + thanosRuler: + spec: + evaluationInterval: 1m