Загрузка данных


dmitriev-aal@VDI-Dmitriev-A:~/Desktop/appfarm/infra/k8s/nexus$ helmfile --environment cpstbl -f deploy/helmfile.yaml -l name=nexus-prometheusrule diff
Decrypting secret /home/dmitriev-aal/Desktop/appfarm/infra/k8s/nexus/deploy/cpstbl/secrets-nexus-scraper.yaml
Decrypting secret /home/dmitriev-aal/Desktop/appfarm/infra/k8s/nexus/deploy/cpstbl/secrets-nexus-cleaner.yaml
Comparing release=nexus-prometheusrule, chart=rshb-charts/prometheus-rules, namespace=nexus
nexus, nexus-prometheusrule-nexus-nexus-alerts, PrometheusRule (monitoring.coreos.com) has changed:
  # Source: prometheus-rules/templates/prometheusrules.yaml
  apiVersion: monitoring.coreos.com/v1
  kind: PrometheusRule
  metadata:
    name: nexus-prometheusrule-nexus-nexus-alerts
    namespace: nexus
    labels:
      app: kube-prometheus-stack
      helm.sh/chart: prometheus-rules-1.1.7
      release: kps
  spec:
    groups:
      - name: nexus-prometheusrule-nexus-nexus-alerts
        rules:
        - alert: NexusPVCFreeSpaceLow
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: PersistentVolume {{ $labels.persistentvolumeclaim }} free space is
              below 20%.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Nexus PVC free space is running low.
          expr: |-
            sum without(instance,node) (
              kubelet_volume_stats_available_bytes{
                job="kubelet",
                metrics_path="/metrics",
                namespace="nexus",
                persistentvolumeclaim="nexus-nexus3-data"
              }
            )
            /
            sum without(instance,node) (
              kubelet_volume_stats_capacity_bytes{
                job="kubelet",
                metrics_path="/metrics",
                namespace="nexus",
                persistentvolumeclaim="nexus-nexus3-data"
              }
            )
            < 0.20
          for: 5m
          labels:
            component: storage
            environment: production
            service: nexus
            severity: warning
            team: sre
        - alert: NexusPVCFreeSpaceCritical
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: PersistentVolume {{ $labels.persistentvolumeclaim }} free space is
              below 10%.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Nexus PVC free space is critically low.
          expr: |-
            sum without(instance,node) (
              kubelet_volume_stats_available_bytes{
                job="kubelet",
                metrics_path="/metrics",
                namespace="nexus",
                persistentvolumeclaim="nexus-nexus3-data"
              }
            )
            /
            sum without(instance,node) (
              kubelet_volume_stats_capacity_bytes{
                job="kubelet",
                metrics_path="/metrics",
                namespace="nexus",
                persistentvolumeclaim="nexus-nexus3-data"
              }
            )
            < 0.10
          for: 3m
          labels:
            component: storage
            environment: production
            service: nexus
            severity: critical
            team: sre
        - alert: NexusPVCInodesLow
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: PersistentVolume {{ $labels.persistentvolumeclaim }} inode usage
              is above 90%.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Nexus PVC inode count is running low.
          expr: |-
            kubelet_volume_stats_inodes_free{
              job="kubelet",
              metrics_path="/metrics",
              namespace="nexus",
              persistentvolumeclaim="nexus-nexus3-data"
            }
            /
            kubelet_volume_stats_inodes{
              job="kubelet",
              metrics_path="/metrics",
              namespace="nexus",
              persistentvolumeclaim="nexus-nexus3-data"
            }
            < 0.10
          for: 5m
          labels:
            component: storage
            environment: production
            service: nexus
            severity: warning
            team: sre
        - alert: NexusTaskFailures
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Task with name {{ $labels.name }} was completed with NOT OK status
              {{ $labels.lastRunResult }}.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Check Nexus UI (System -> Tasks) and logs for investigating.
          expr: nexus_tasks_status{lastRunResult!~"OK|<nil>"} > 0
          for: 1m
          labels:
            component: tasks
            environment: production
            service: nexus
            severity: critical
            team: sre
nexus, nexus-prometheusrule-nexus-nexus-blackbox-proxy, PrometheusRule (monitoring.coreos.com) has changed:
  # Source: prometheus-rules/templates/prometheusrules.yaml
  apiVersion: monitoring.coreos.com/v1
  kind: PrometheusRule
  metadata:
    name: nexus-prometheusrule-nexus-nexus-blackbox-proxy
    namespace: nexus
    labels:
      app: kube-prometheus-stack
      helm.sh/chart: prometheus-rules-1.1.7
      release: kps
  spec:
    groups:
      - name: nexus-prometheusrule-nexus-nexus-blackbox-proxy
        rules:
        - alert: NexusProxyTargetDown
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Blackbox probe through Nexus proxy failed for {{ $labels.instance
              }} for 3 minutes.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Proxy target {{ $labels.instance }} is down.
          expr: probe_success{job="nexus-proxy-targets"} == 0
          for: 3m
          labels:
            component: proxy
            environment: production
            service: nexus
            severity: critical
            team: sre
        - alert: NexusProxyHighLatency
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Proxy request to {{ $labels.instance }} has p95 latency above 3 seconds
              for 5 minutes.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: High Nexus proxy latency for {{ $labels.instance }}.
          expr: quantile_over_time(0.95, probe_duration_seconds{job="nexus-proxy-targets"}[5m])
            > 3
          for: 5m
          labels:
            component: proxy
            environment: production
            service: nexus
            severity: warning
            team: sre
        - alert: NexusProxyCriticalLatency
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Proxy request to {{ $labels.instance }} has p99 latency above 5 seconds
              for 3 minutes.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Critical Nexus proxy latency for {{ $labels.instance }}.
          expr: quantile_over_time(0.99, probe_duration_seconds{job="nexus-proxy-targets"}[5m])
            > 5
          for: 3m
          labels:
            component: proxy
            environment: production
            service: nexus
            severity: critical
            team: sre
        - alert: NexusProxyCertExpiringSoon
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: SSL certificate for {{ $labels.instance }} will expire in {{ $value
              | humanizeDuration }}.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: SSL certificate expiring soon for {{ $labels.instance }}.
          expr: probe_ssl_earliest_cert_expiry{job="nexus-proxy-targets"} - time() < 86400
            * 30
          for: 1h
          labels:
            component: certificates
            environment: production
            service: nexus
            severity: warning
            team: sre
        - alert: NexusProxyHTTP4xxError
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Proxy request to {{ $labels.instance }} returned HTTP 4xx status
              code {{ $value }}.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: HTTP 4xx status for {{ $labels.instance }}
          expr: probe_http_status_code{job="nexus-proxy-targets"} >= 400 and probe_http_status_code{job="nexus-proxy-targets"}
            < 500
          for: 1m
          labels:
            component: proxy
            environment: production
            service: nexus
            severity: warning
            team: sre
        - alert: NexusProxyHTTP5xxError
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Proxy request to {{ $labels.instance }} returned HTTP 5xx status
              code {{ $value }}.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: HTTP 5xx status for {{ $labels.instance }}
          expr: probe_http_status_code{job="nexus-proxy-targets"} >= 500 and probe_http_status_code{job="nexus-proxy-targets"}
            < 600
          for: 1m
          labels:
            component: proxy
            environment: production
            service: nexus
            severity: warning
            team: sre
nexus, nexus-prometheusrule-nexus-nexus-common-rules, PrometheusRule (monitoring.coreos.com) has changed:
  # Source: prometheus-rules/templates/prometheusrules.yaml
  apiVersion: monitoring.coreos.com/v1
  kind: PrometheusRule
  metadata:
    name: nexus-prometheusrule-nexus-nexus-common-rules
    namespace: nexus
    labels:
      app: kube-prometheus-stack
      helm.sh/chart: prometheus-rules-1.1.7
      release: kps
  spec:
    groups:
      - name: nexus-prometheusrule-nexus-nexus-common-rules
        rules:
        - alert: NexusAvailabilityLow
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Nexus pod readiness is below 95% for 3 minutes.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Nexus availability is low.
          expr: avg(avg_over_time(kube_pod_container_status_ready{namespace="nexus",container="nexus3"}[5m]))
            < 0.95
          for: 3m
          labels:
            component: availability
            environment: production
            service: nexus
            severity: critical
            team: sre
        - alert: NexusIngress4xxHigh
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Nexus 4xx ratio is above 2% for 3 minutes.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Nexus 4xx error ratio is high.
          expr: |-
            sum(rate(org_eclipse_jetty_webapp_WebAppContext_4xx_responses_total{namespace="nexus"}[10m]))
              /
            (
              sum(rate(org_eclipse_jetty_webapp_WebAppContext_2xx_responses_total{namespace="nexus"}[10m]))
              +
              sum(rate(org_eclipse_jetty_webapp_WebAppContext_4xx_responses_total{namespace="nexus"}[10m]))
              +
              sum(rate(org_eclipse_jetty_webapp_WebAppContext_5xx_responses_total{namespace="nexus"}[10m]))
            ) > 0.02
          for: 3m
          labels:
            component: ingress
            environment: production
            service: nexus
            severity: warning
            team: sre
        - alert: NexusIngress4xxCritical
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Nexus 4xx ratio is above 5% for 3 minutes.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Nexus 4xx error ratio is critical.
          expr: |-
            sum(rate(org_eclipse_jetty_webapp_WebAppContext_4xx_responses_total{namespace="nexus"}[10m]))
              /
            (
              sum(rate(org_eclipse_jetty_webapp_WebAppContext_2xx_responses_total{namespace="nexus"}[10m]))
              +
              sum(rate(org_eclipse_jetty_webapp_WebAppContext_4xx_responses_total{namespace="nexus"}[10m]))
              +
              sum(rate(org_eclipse_jetty_webapp_WebAppContext_5xx_responses_total{namespace="nexus"}[10m]))
            ) > 0.05
          for: 3m
          labels:
            component: ingress
            environment: production
            service: nexus
            severity: critical
            team: sre
        - alert: NexusIngress5xxHigh
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Nexus 5xx ratio is above 2% for 3 minutes.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Nexus 5xx error ratio is high.
          expr: |-
            sum(rate(org_eclipse_jetty_webapp_WebAppContext_5xx_responses_total{namespace="nexus"}[10m]))
              /
            (
              sum(rate(org_eclipse_jetty_webapp_WebAppContext_2xx_responses_total{namespace="nexus"}[10m]))
              +
              sum(rate(org_eclipse_jetty_webapp_WebAppContext_4xx_responses_total{namespace="nexus"}[10m]))
              +
              sum(rate(org_eclipse_jetty_webapp_WebAppContext_5xx_responses_total{namespace="nexus"}[10m]))
            ) > 0.02
          for: 3m
          labels:
            component: ingress
            environment: production
            service: nexus
            severity: warning
            team: sre
        - alert: NexusIngress5xxCritical
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Nexus 5xx ratio is above 5% for 3 minutes.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Nexus 5xx error ratio is critical.
          expr: |-
            sum(rate(org_eclipse_jetty_webapp_WebAppContext_5xx_responses_total{namespace="nexus"}[10m]))
              /
            (
              sum(rate(org_eclipse_jetty_webapp_WebAppContext_2xx_responses_total{namespace="nexus"}[10m]))
              +
              sum(rate(org_eclipse_jetty_webapp_WebAppContext_4xx_responses_total{namespace="nexus"}[10m]))
              +
              sum(rate(org_eclipse_jetty_webapp_WebAppContext_5xx_responses_total{namespace="nexus"}[10m]))
            ) > 0.05
          for: 3m
          labels:
            component: ingress
            environment: production
            service: nexus
            severity: critical
            team: sre
        - alert: NexusRegistryLatencyHigh
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Nexus repository read p99 latency is above 3 seconds for 5 minutes.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Nexus repository read latency is high.
          expr: org_sonatype_nexus_coreui_RepositoryComponent_read_timer{quantile="0.99",namespace="nexus"}
            > 3
          for: 5m
          labels:
            component: registry
            environment: production
            service: nexus
            severity: warning
            team: sre
        - alert: NexusRegistryLatencyCritical
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Nexus repository read p99 latency is above 5 seconds for 3 minutes.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Nexus repository read latency is critical.
          expr: org_sonatype_nexus_coreui_RepositoryComponent_read_timer{quantile="0.99",namespace="nexus"}
            > 5
          for: 3m
          labels:
            component: registry
            environment: production
            service: nexus
            severity: critical
            team: sre
        - alert: NexusBlobstoreUnavailable
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Nexus default blobstore has no available space.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Nexus blobstore is unavailable.
          expr: nexus_blobstores_stats_availableSpaceInBytes{namespace="nexus", name="default"}
            == 0
          for: 1m
          labels:
            component: blobstore
            environment: production
            service: nexus
            severity: critical
            team: sre
        - alert: NexusJvmHeapHigh
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Nexus JVM heap usage is above 90% for 3 minutes.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Nexus JVM heap usage is high.
          expr: |-
            sum(jvm_memory_heap_used{namespace="nexus", container="nexus3"}) by(pod)
              /
            sum(jvm_memory_heap_max{namespace="nexus", container="nexus3"}) by(pod)
              > 0.9
          for: 3m
          labels:
            component: jvm
            environment: production
            service: nexus
            severity: critical
            team: sre
        - alert: NexusCpuThrottlingHigh
          annotations:
+           dashboard_url: https://grafana.cpstbl.rshbdev.ru/d/H7vCRCJGk/nexus
            description: Nexus CPU throttling ratio is above 20% for 5 minutes.
+           runbook_url: https://gitlab.rshbdev.ru/appfarm/infra/k8s/nexus/-/tree/master/deploy/runbook/runbook.md
            summary: Nexus CPU throttling is high.
          expr: |-
            rate(container_cpu_cfs_throttled_periods_total{namespace="nexus",container="nexus3"}[5m])
              /
            rate(container_cpu_cfs_periods_total{namespace="nexus",container="nexus3"}[5m])
              > 0.2
          for: 5m
          labels:
            component: cpu
            environment: production
            service: nexus
            severity: warning
            team: sre

dmitriev-aal@VDI-Dmitriev-A:~/Desktop/appfarm/infra/k8s/nexus$