Alerts


/etc/config/alerting_rules.yml > BlackBox Alerts
Probe failure (0 active)
alert: Probe failure
expr: avg_over_time(probe_success{job=~"compliance-service-check|dpm-service-check",namespace!="kube-system"}[15m]) < 0.5
for: 5m
annotations:
  summary: The service {{ $labels.job }} is unreachable or down. please check the cluster for further information.
Public endpoint check (0 active)
alert: Public endpoint check
expr: probe_success{job=~"external.*"} == 0
for: 1m
labels:
  severity: warning
annotations:
  summary: The service {{ $labels.job }} is unreachble from internet, please check if URL is pointing to public endpoint.
/etc/config/alerting_rules.yml > MSSQL Alerts
KubernetesPodNotHealthy (81 active)
alert: KubernetesPodNotHealthy
expr: min_over_time(sum by(namespace, pod) (kube_pod_status_phase{namespace!="kube-system",phase=~"Pending|Unknown|Failed"})[15m:1m]) > 0
labels:
  severity: critical
annotations:
  description: |-
    Pod has been in a non-ready state for longer than 15 minutes.
      VALUE = {{ $value }}
      LABELS = {{ $labels }}
  summary: Kubernetes Pod not healthy (instance {{ $labels.pod }})
Labels State Active Since Value
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-rr848" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="prometheus-query-exporter-66b8dc9cb4-jgmvz" severity="critical" firing 2026-07-17 05:53:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-mx8rd" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-crg4d" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-7jqgl" severity="critical" firing 2026-07-21 22:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-smqn8" severity="critical" firing 2026-07-21 22:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="prometheus-query-exporter-66b8dc9cb4-srjn8" severity="critical" firing 2026-07-17 05:53:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-p2gz9" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="mongosqld-64988b4445-8vg9g" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="prometheus-query-exporter-66b8dc9cb4-hdnp2" severity="critical" firing 2026-07-17 05:53:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="mongosqld-64988b4445-4bkv7" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="mongosqld-64988b4445-cvp9t" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-4jp45" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-ph6gp" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-8bm5l" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-mnfdh" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="svc-reporting-7f45f6bcdc-fblfm" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-84bkx" severity="critical" firing 2026-07-21 22:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="svc-reporting-7f45f6bcdc-27d4p" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="mongosqld-64988b4445-bcnl2" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="mongosqld-64988b4445-jfjs6" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-sbcvz" severity="critical" firing 2026-07-15 05:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="prometheus-query-exporter-66b8dc9cb4-b99p4" severity="critical" firing 2026-07-17 05:53:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="prometheus-query-exporter-66b8dc9cb4-tq7vz" severity="critical" firing 2026-07-17 05:53:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="audit-metrics-kafka-store-65b7f9c8d-mj7jl" severity="critical" firing 2026-07-13 00:52:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="job-licensestatuscheck-0.0.9.228.0-774q8" severity="critical" firing 2026-07-26 00:15:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="mongosqld-64988b4445-qxbzb" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-nhwkj" severity="critical" firing 2026-07-21 22:06:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-kj6m4" severity="critical" firing 2026-07-21 22:20:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-txwp9" severity="critical" firing 2026-07-21 22:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="prometheus-query-exporter-66b8dc9cb4-hdp79" severity="critical" firing 2026-07-17 05:53:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-rvckz" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-nmgr4" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-pwzjf" severity="critical" firing 2026-07-21 22:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="svc-reporting-7f45f6bcdc-5v5x9" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-lpddt" severity="critical" firing 2026-07-21 22:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-ckrg7" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="mongosqld-64988b4445-hsgc7" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-w9bfn" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-m72v4" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="svc-reporting-7f45f6bcdc-b89wm" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-mqgqk" severity="critical" firing 2026-07-19 10:13:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-955cl" severity="critical" firing 2026-07-21 22:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="audit-metrics-kafka-store-65b7f9c8d-v6hp6" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="mongosqld-64988b4445-tvf8b" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-4cgf7" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-wlc25" severity="critical" firing 2026-07-21 22:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-vwn6h" severity="critical" firing 2026-07-17 15:30:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-jj9t4" severity="critical" firing 2026-07-21 22:06:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="svc-reporting-7f45f6bcdc-w8zg5" severity="critical" firing 2026-07-13 00:53:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-r54b9" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="svc-reporting-7f45f6bcdc-qwcvv" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="mongosqld-64988b4445-6bxdl" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-j5zhv" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-gb74m" severity="critical" firing 2026-07-21 22:06:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="prometheus-query-exporter-66b8dc9cb4-6jv5r" severity="critical" firing 2026-07-17 06:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="prometheus-query-exporter-66b8dc9cb4-9qhrs" severity="critical" firing 2026-07-17 05:53:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="prometheus-query-exporter-66b8dc9cb4-jt7d7" severity="critical" firing 2026-07-17 05:53:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-mjq8j" severity="critical" firing 2026-07-21 22:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="prometheus-query-exporter-66b8dc9cb4-7m57b" severity="critical" firing 2026-07-13 01:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-clb4k" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-7rh2p" severity="critical" firing 2026-07-21 22:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-cnvdz" severity="critical" firing 2026-07-13 01:12:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-tps4v" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="svc-reporting-7f45f6bcdc-kzk5k" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-xznrz" severity="critical" firing 2026-07-21 22:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="prometheus-query-exporter-66b8dc9cb4-92x96" severity="critical" firing 2026-07-17 05:53:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-mlcrq" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="mongosqld-64988b4445-lv92t" severity="critical" firing 2026-07-13 00:52:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-l7zbn" severity="critical" firing 2026-07-21 22:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="svc-reporting-7f45f6bcdc-hwsf5" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="svc-reporting-7f45f6bcdc-msf5h" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-tz2ws" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-v9t6n" severity="critical" firing 2026-07-17 15:16:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="svc-reporting-7f45f6bcdc-sqvc4" severity="critical" firing 2026-07-13 00:52:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-sfm9j" severity="critical" firing 2026-07-21 22:07:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="ingress-nginx-controller-6599f9f5fb-rrzwj" severity="critical" firing 2026-07-23 18:49:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="monitoring" pod="prometheus-query-exporter-66b8dc9cb4-p5fcj" severity="critical" firing 2026-07-17 05:53:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="mongosqld-64988b4445-n4sdt" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="mongosqld-64988b4445-kw6pf" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
alertname="KubernetesPodNotHealthy" namespace="p438" pod="mongosqld-64988b4445-9tj7p" severity="critical" firing 2026-07-13 00:58:46.936640537 +0000 UTC 1
DatabaseMaintainenceJobCountIncreased (0 active)
alert: DatabaseMaintainenceJobCountIncreased
expr: jobcount{job="prometheus-query-exporter"} > 0
for: 1m
annotations:
  description: Database Maintainence Job count for database
  summary: Database Maintainence Job count for database
HostHighCpuLoad (0 active)
alert: HostHighCpuLoad
expr: 100 - (avg by(instance) (rate(node_cpu_seconds_total{mode="idle"}[2m])) * 100) > 80
for: 2m
labels:
  severity: warning
annotations:
  description: |-
    CPU load is > 80%
      VALUE = {{ $value }}
      LABELS = {{ $labels }}
  summary: Host high CPU load (instance {{ $labels.pod }})
KubernetesDiskPressure (0 active)
alert: KubernetesDiskPressure
expr: kube_node_status_condition{condition="DiskPressure",namespace!="kube-system",status="true"} == 1
for: 2m
labels:
  severity: critical
annotations:
  description: |-
    {{ $labels.node }} has DiskPressure condition
      VALUE = {{ $value }}
      LABELS = {{ $labels }}
  summary: Kubernetes disk pressure (instance {{ $labels.pod }})
KubernetesMemoryPressure (0 active)
alert: KubernetesMemoryPressure
expr: kube_node_status_condition{condition="MemoryPressure",namespace!="kube-system",status="true"} == 1
for: 2m
labels:
  severity: critical
annotations:
  description: |-
    {{ $labels.node }} has MemoryPressure condition
      VALUE = {{ $value }}
      LABELS = {{ $labels }}
  summary: Kubernetes memory pressure (instance {{ $labels.pod }})
KubernetesNodeReady (0 active)
alert: KubernetesNodeReady
expr: kube_node_status_condition{condition="Ready",namespace!="kube-system",status="true"} == 0
for: 10m
labels:
  severity: critical
annotations:
  description: |-
    Node {{ $labels.node }} has been unready for a long time
      VALUE = {{ $value }}
      LABELS = {{ $labels }}
  summary: Kubernetes Node ready (instance {{ $labels.pod }})
KubernetesOutOfCapacity (0 active)
alert: KubernetesOutOfCapacity
expr: sum by(node) ((kube_pod_status_phase{namespace!="kube-system",phase="Running"} == 1) + on(pod, namespace) group_left(node) (0 * kube_pod_info)) / sum by(node) (kube_node_status_allocatable_pods{namespace!="kube-system"}) * 100 > 90
for: 2m
labels:
  severity: warning
annotations:
  description: |-
    {{ $labels.node }} is out of capacity
      VALUE = {{ $value }}
      LABELS = {{ $labels }}
  summary: Kubernetes out of capacity (instance {{ $labels.pod }})
KubernetesOutOfDisk (0 active)
alert: KubernetesOutOfDisk
expr: kube_node_status_condition{condition="OutOfDisk",namespace!="kube-system",status="true"} == 1
for: 2m
labels:
  severity: critical
annotations:
  description: |-
    {{ $labels.node }} has OutOfDisk condition
      VALUE = {{ $value }}
      LABELS = {{ $labels }}
  summary: Kubernetes out of disk (instance {{ $labels.pod }})
KubernetesPersistentvolumeError (0 active)
alert: KubernetesPersistentvolumeError
expr: kube_persistentvolume_status_phase{job="kube-state-metrics",namespace!="kube-system",phase=~"Failed|Pending"} > 0
labels:
  severity: critical
annotations:
  description: |-
    Persistent volume is in bad state
      VALUE = {{ $value }}
      LABELS = {{ $labels }}
  summary: Kubernetes PersistentVolume error (instance {{ $labels.pod }})
KubernetesPersistentvolumeclaimPending (0 active)
alert: KubernetesPersistentvolumeclaimPending
expr: kube_persistentvolumeclaim_status_phase{namespace!="kube-system",phase="Pending"} == 1
for: 2m
labels:
  severity: warning
annotations:
  description: |-
    PersistentVolumeClaim {{ $labels.namespace }}/{{ $labels.persistentvolumeclaim }} is pending
      VALUE = {{ $value }}
      LABELS = {{ $labels }}
  summary: Kubernetes PersistentVolumeClaim pending (instance {{ $labels.pod }})
KubernetesPodCrashLooping (0 active)
alert: KubernetesPodCrashLooping
expr: increase(kube_pod_container_status_restarts_total{namespace!="kube-system"}[1m]) > 3
for: 2m
labels:
  severity: warning
annotations:
  description: |-
    Pod {{ $labels.pod }} is crash looping
      VALUE = {{ $value }}
      LABELS = {{ $labels }}
  summary: Kubernetes pod crash looping (instance {{ $labels.pod }})
KubernetesVolumeOutOfDiskSpace (0 active)
MSSQL connectivity alert (0 active)
alert: MSSQL connectivity alert
expr: up{job="prometheus-mssql-exporter"} == 0
for: 1m
labels:
  severity: Critical
annotations:
  summary: The service {{ $labels.job }} is unreachable or down. please check the MSSQL for further information.
compliance alert (0 active)
alert: compliance alert
expr: probe_success{job="compliance",namespace!="kube-system"} == 1
labels:
  severity: warning
annotations:
  summary: The service {{ $labels.job }} compliance is enabled.
compliance alert (0 active)
alert: compliance alert
expr: probe_success{job="compliance",namespace!="kube-system"} == 0
labels:
  Notification: None
  severity: warning
annotations:
  summary: The service {{ $labels.job }} compliance is disabled.