Загрузка данных
dmitriev-aal@VDI-Dmitriev-A:~/Desktop/appfarm/infra/k8s/nexus$ helmfile --environment cpstbl -f deploy/helmfile.yaml -l name=nexus-prometheusrule diff
Decrypting secret /home/dmitriev-aal/Desktop/appfarm/infra/k8s/nexus/deploy/cpstbl/secrets-nexus-scraper.yaml
Decrypting secret /home/dmitriev-aal/Desktop/appfarm/infra/k8s/nexus/deploy/cpstbl/secrets-nexus-cleaner.yaml
Comparing release=nexus-prometheusrule, chart=rshb-charts/prometheus-rules, namespace=nexus
nexus, nexus-prometheusrule-nexus-nexus-alerts, PrometheusRule (monitoring.coreos.com) has changed:
# Source: prometheus-rules/templates/prometheusrules.yaml
apiVersion: monitoring.coreos.com/v1
kind: PrometheusRule
metadata:
name: nexus-prometheusrule-nexus-nexus-alerts
namespace: nexus
labels:
app: kube-prometheus-stack
helm.sh/chart: prometheus-rules-1.1.7
release: kps
spec:
groups:
- name: nexus-prometheusrule-nexus-nexus-alerts
rules:
- alert: NexusPVCFreeSpaceLow
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: PersistentVolume {{ $labels.persistentvolumeclaim }} free space is
below 20%.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Nexus PVC free space is running low.
expr: |-
sum without(instance,node) (
kubelet_volume_stats_available_bytes{
job="kubelet",
metrics_path="/metrics",
namespace="nexus",
persistentvolumeclaim="nexus-nexus3-data"
}
)
/
sum without(instance,node) (
kubelet_volume_stats_capacity_bytes{
job="kubelet",
metrics_path="/metrics",
namespace="nexus",
persistentvolumeclaim="nexus-nexus3-data"
}
)
< 0.20
for: 5m
labels:
component: storage
environment: production
service: nexus
severity: warning
team: sre
- alert: NexusPVCFreeSpaceCritical
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: PersistentVolume {{ $labels.persistentvolumeclaim }} free space is
below 10%.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Nexus PVC free space is critically low.
expr: |-
sum without(instance,node) (
kubelet_volume_stats_available_bytes{
job="kubelet",
metrics_path="/metrics",
namespace="nexus",
persistentvolumeclaim="nexus-nexus3-data"
}
)
/
sum without(instance,node) (
kubelet_volume_stats_capacity_bytes{
job="kubelet",
metrics_path="/metrics",
namespace="nexus",
persistentvolumeclaim="nexus-nexus3-data"
}
)
< 0.10
for: 3m
labels:
component: storage
environment: production
service: nexus
severity: critical
team: sre
- alert: NexusPVCInodesLow
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: PersistentVolume {{ $labels.persistentvolumeclaim }} inode usage
is above 90%.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Nexus PVC inode count is running low.
expr: |-
kubelet_volume_stats_inodes_free{
job="kubelet",
metrics_path="/metrics",
namespace="nexus",
persistentvolumeclaim="nexus-nexus3-data"
}
/
kubelet_volume_stats_inodes{
job="kubelet",
metrics_path="/metrics",
namespace="nexus",
persistentvolumeclaim="nexus-nexus3-data"
}
< 0.10
for: 5m
labels:
component: storage
environment: production
service: nexus
severity: warning
team: sre
- alert: NexusTaskFailures
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Task with name {{ $labels.name }} was completed with NOT OK status
{{ $labels.lastRunResult }}.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Check Nexus UI (System -> Tasks) and logs for investigating.
expr: nexus_tasks_status{lastRunResult!~"OK|<nil>"} > 0
for: 1m
labels:
component: tasks
environment: production
service: nexus
severity: critical
team: sre
nexus, nexus-prometheusrule-nexus-nexus-blackbox-proxy, PrometheusRule (monitoring.coreos.com) has changed:
# Source: prometheus-rules/templates/prometheusrules.yaml
apiVersion: monitoring.coreos.com/v1
kind: PrometheusRule
metadata:
name: nexus-prometheusrule-nexus-nexus-blackbox-proxy
namespace: nexus
labels:
app: kube-prometheus-stack
helm.sh/chart: prometheus-rules-1.1.7
release: kps
spec:
groups:
- name: nexus-prometheusrule-nexus-nexus-blackbox-proxy
rules:
- alert: NexusProxyTargetDown
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Blackbox probe through Nexus proxy failed for {{ $labels.instance
}} for 3 minutes.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Proxy target {{ $labels.instance }} is down.
expr: probe_success{job="nexus-proxy-targets"} == 0
for: 3m
labels:
component: proxy
environment: production
service: nexus
severity: critical
team: sre
- alert: NexusProxyHighLatency
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Proxy request to {{ $labels.instance }} has p95 latency above 3 seconds
for 5 minutes.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: High Nexus proxy latency for {{ $labels.instance }}.
expr: quantile_over_time(0.95, probe_duration_seconds{job="nexus-proxy-targets"}[5m])
> 3
for: 5m
labels:
component: proxy
environment: production
service: nexus
severity: warning
team: sre
- alert: NexusProxyCriticalLatency
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Proxy request to {{ $labels.instance }} has p99 latency above 5 seconds
for 3 minutes.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Critical Nexus proxy latency for {{ $labels.instance }}.
expr: quantile_over_time(0.99, probe_duration_seconds{job="nexus-proxy-targets"}[5m])
> 5
for: 3m
labels:
component: proxy
environment: production
service: nexus
severity: critical
team: sre
- alert: NexusProxyCertExpiringSoon
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: SSL certificate for {{ $labels.instance }} will expire in {{ $value
| humanizeDuration }}.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: SSL certificate expiring soon for {{ $labels.instance }}.
expr: probe_ssl_earliest_cert_expiry{job="nexus-proxy-targets"} - time() < 86400
* 30
for: 1h
labels:
component: certificates
environment: production
service: nexus
severity: warning
team: sre
- alert: NexusProxyHTTP4xxError
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Proxy request to {{ $labels.instance }} returned HTTP 4xx status
code {{ $value }}.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: HTTP 4xx status for {{ $labels.instance }}
expr: probe_http_status_code{job="nexus-proxy-targets"} >= 400 and probe_http_status_code{job="nexus-proxy-targets"}
< 500
for: 1m
labels:
component: proxy
environment: production
service: nexus
severity: warning
team: sre
- alert: NexusProxyHTTP5xxError
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Proxy request to {{ $labels.instance }} returned HTTP 5xx status
code {{ $value }}.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: HTTP 5xx status for {{ $labels.instance }}
expr: probe_http_status_code{job="nexus-proxy-targets"} >= 500 and probe_http_status_code{job="nexus-proxy-targets"}
< 600
for: 1m
labels:
component: proxy
environment: production
service: nexus
severity: warning
team: sre
nexus, nexus-prometheusrule-nexus-nexus-common-rules, PrometheusRule (monitoring.coreos.com) has changed:
# Source: prometheus-rules/templates/prometheusrules.yaml
apiVersion: monitoring.coreos.com/v1
kind: PrometheusRule
metadata:
name: nexus-prometheusrule-nexus-nexus-common-rules
namespace: nexus
labels:
app: kube-prometheus-stack
helm.sh/chart: prometheus-rules-1.1.7
release: kps
spec:
groups:
- name: nexus-prometheusrule-nexus-nexus-common-rules
rules:
- alert: NexusAvailabilityLow
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Nexus pod readiness is below 95% for 3 minutes.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Nexus availability is low.
expr: avg(avg_over_time(kube_pod_container_status_ready{namespace="nexus",container="nexus3"}[5m]))
< 0.95
for: 3m
labels:
component: availability
environment: production
service: nexus
severity: critical
team: sre
- alert: NexusIngress4xxHigh
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Nexus 4xx ratio is above 2% for 3 minutes.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Nexus 4xx error ratio is high.
expr: |-
sum(rate(org_eclipse_jetty_webapp_WebAppContext_4xx_responses_total{namespace="nexus"}[10m]))
/
(
sum(rate(org_eclipse_jetty_webapp_WebAppContext_2xx_responses_total{namespace="nexus"}[10m]))
+
sum(rate(org_eclipse_jetty_webapp_WebAppContext_4xx_responses_total{namespace="nexus"}[10m]))
+
sum(rate(org_eclipse_jetty_webapp_WebAppContext_5xx_responses_total{namespace="nexus"}[10m]))
) > 0.02
for: 3m
labels:
component: ingress
environment: production
service: nexus
severity: warning
team: sre
- alert: NexusIngress4xxCritical
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Nexus 4xx ratio is above 5% for 3 minutes.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Nexus 4xx error ratio is critical.
expr: |-
sum(rate(org_eclipse_jetty_webapp_WebAppContext_4xx_responses_total{namespace="nexus"}[10m]))
/
(
sum(rate(org_eclipse_jetty_webapp_WebAppContext_2xx_responses_total{namespace="nexus"}[10m]))
+
sum(rate(org_eclipse_jetty_webapp_WebAppContext_4xx_responses_total{namespace="nexus"}[10m]))
+
sum(rate(org_eclipse_jetty_webapp_WebAppContext_5xx_responses_total{namespace="nexus"}[10m]))
) > 0.05
for: 3m
labels:
component: ingress
environment: production
service: nexus
severity: critical
team: sre
- alert: NexusIngress5xxHigh
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Nexus 5xx ratio is above 2% for 3 minutes.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Nexus 5xx error ratio is high.
expr: |-
sum(rate(org_eclipse_jetty_webapp_WebAppContext_5xx_responses_total{namespace="nexus"}[10m]))
/
(
sum(rate(org_eclipse_jetty_webapp_WebAppContext_2xx_responses_total{namespace="nexus"}[10m]))
+
sum(rate(org_eclipse_jetty_webapp_WebAppContext_4xx_responses_total{namespace="nexus"}[10m]))
+
sum(rate(org_eclipse_jetty_webapp_WebAppContext_5xx_responses_total{namespace="nexus"}[10m]))
) > 0.02
for: 3m
labels:
component: ingress
environment: production
service: nexus
severity: warning
team: sre
- alert: NexusIngress5xxCritical
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Nexus 5xx ratio is above 5% for 3 minutes.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Nexus 5xx error ratio is critical.
expr: |-
sum(rate(org_eclipse_jetty_webapp_WebAppContext_5xx_responses_total{namespace="nexus"}[10m]))
/
(
sum(rate(org_eclipse_jetty_webapp_WebAppContext_2xx_responses_total{namespace="nexus"}[10m]))
+
sum(rate(org_eclipse_jetty_webapp_WebAppContext_4xx_responses_total{namespace="nexus"}[10m]))
+
sum(rate(org_eclipse_jetty_webapp_WebAppContext_5xx_responses_total{namespace="nexus"}[10m]))
) > 0.05
for: 3m
labels:
component: ingress
environment: production
service: nexus
severity: critical
team: sre
- alert: NexusRegistryLatencyHigh
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Nexus repository read p99 latency is above 3 seconds for 5 minutes.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Nexus repository read latency is high.
expr: org_sonatype_nexus_coreui_RepositoryComponent_read_timer{quantile="0.99",namespace="nexus"}
> 3
for: 5m
labels:
component: registry
environment: production
service: nexus
severity: warning
team: sre
- alert: NexusRegistryLatencyCritical
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Nexus repository read p99 latency is above 5 seconds for 3 minutes.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Nexus repository read latency is critical.
expr: org_sonatype_nexus_coreui_RepositoryComponent_read_timer{quantile="0.99",namespace="nexus"}
> 5
for: 3m
labels:
component: registry
environment: production
service: nexus
severity: critical
team: sre
- alert: NexusBlobstoreUnavailable
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Nexus default blobstore has no available space.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Nexus blobstore is unavailable.
expr: nexus_blobstores_stats_availableSpaceInBytes{namespace="nexus", name="default"}
== 0
for: 1m
labels:
component: blobstore
environment: production
service: nexus
severity: critical
team: sre
- alert: NexusJvmHeapHigh
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Nexus JVM heap usage is above 90% for 3 minutes.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Nexus JVM heap usage is high.
expr: |-
sum(jvm_memory_heap_used{namespace="nexus", container="nexus3"}) by(pod)
/
sum(jvm_memory_heap_max{namespace="nexus", container="nexus3"}) by(pod)
> 0.9
for: 3m
labels:
component: jvm
environment: production
service: nexus
severity: critical
team: sre
- alert: NexusCpuThrottlingHigh
annotations:
+ dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
description: Nexus CPU throttling ratio is above 20% for 5 minutes.
+ runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
summary: Nexus CPU throttling is high.
expr: |-
rate(container_cpu_cfs_throttled_periods_total{namespace="nexus",container="nexus3"}[5m])
/
rate(container_cpu_cfs_periods_total{namespace="nexus",container="nexus3"}[5m])
> 0.2
for: 5m
labels:
component: cpu
environment: production
service: nexus
severity: warning
team: sre
dmitriev-aal@VDI-Dmitriev-A:~/Desktop/appfarm/infra/k8s/nexus$