PageSourceSearch

https://www.rabbitmq.com/assets/js/84f1ad3c.a09e6b5f.js

js rabbitmq.com collected 2026-09-24 06:06:01 UTC 18,946 bytes, 1 lines download raw bytes

1"use strict";(self.webpackChunkrabbitmq_website=self.webpackChunkrabbitmq_website||[]).push([["17182"],{86103(e,t,r){r.r(t),r.d(t,{assets:()=>l,contentTitle:()=>o,default:()=>b,frontMatter:()=>n,metadata:()=>s,toc:()=>h});var s=r(94308),a=r(74848),i=r(28453);let n={title:"Notify me when RabbitMQ has a problem",tags:["Kubernetes","New Features"],authors:["dansari","glazu"]},o,l={authorsImageUrls:[void 0,void 0]},h=[{value:"What alerts are available today?",id:"what-alerts-are-available-today",level:2},{value:"How to get started quickly?",id:"how-to-get-started-quickly",level:2},{value:"Trigger your first RabbitMQ alert",id:"trigger-your-first-rabbitmq-alert",level:2},{value:"Past and current RabbitMQ alerts",id:"past-and-current-rabbitmq-alerts",level:2},{value:"How can you help?",id:"how-can-you-help",level:2}];function c(e){let t={a:"a",code:"code",h2:"h2",img:"img",li:"li",ol:"ol",p:"p",pre:"pre",strong:"strong",ul:"ul",...(0,i.R)(),...e.components};return(0,a.jsxs)(a.Fragment,{children:[(0,a.jsxs)(t.p,{children:["If you want to be notified when your RabbitMQ deployments have a problem, now you can set up the RabbitMQ monitoring and alerting that we have made available in the ",(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/tree/v1.7.0/observability",children:"RabbitMQ Cluster Operator"})," repository.\nRather than asking you to follow a series of steps for setting up RabbitMQ monitoring & alerting, we have combined this in ",(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/blob/v1.7.0/observability/quickstart.sh",children:"a single command"}),".\nWhile this is a Kubernetes-specific quick-start, and you can use these Prometheus alerts outside of Kubernetes, the setup will require more consideration and effort on your part.\nWe share the quick & easy approach, open source and free for all."]}),"\n",(0,a.jsx)(t.p,{children:"When everything is set up and there is a problem with RabbitMQ, this is an example of a notification that you can expect:"}),"\n",(0,a.jsx)(t.p,{children:(0,a.jsx)(t.img,{src:r(92631).A+"",width:"839",height:"707"})}),"\n",(0,a.jsx)(t.p,{children:"The above is a good example of a problem that may not be obvious when it happens, and takes a few steps to troubleshoot.\nRather than losing messages due to a misconfiguration, this notification makes it clear when incoming messages are not routed within RabbitMQ."}),"\n",(0,a.jsx)(t.h2,{id:"what-alerts-are-available-today",children:"What alerts are available today?"}),"\n",(0,a.jsxs)(t.ul,{children:["\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/blob/v1.7.0/observability/prometheus/rules/rabbitmq/no-majority-of-nodes-ready.yml",children:"NoMajorityOfNodesReady"})}),": Only a minority of RabbitMQ nodes can service clients. Some queues are likely to be unavailable, including replicated ones."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/blob/v1.7.0/observability/prometheus/rules/rabbitmq/persistent-volume-missing.yml",children:"PersistentVolumeMissing"})}),": A RabbitMQ node is missing a volume for persisting data and can't boot. This is either a misconfiguration or capacity issue."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/blob/v1.7.0/observability/prometheus/rules/rabbitmq/insufficient-established-erlang-distribution-links.yml",children:"InsufficientEstablishedErlangDistributionLinks"})}),": RabbitMQ nodes are not clustered due to networking issues or incorrect permissions."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/blob/v1.7.0/observability/prometheus/rules/rabbitmq/unroutable-messages.yml",children:"UnroutableMessages"})}),": Messages are not routed from channels to queues. Routing topology needs reviewing."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/blob/v1.7.0/observability/prometheus/rules/rabbitmq/high-connection-churn.yml",children:"HighConnectionChurn"})}),": Clients open and close connections too often, which is an anti-pattern. Clients should use long-lived connections."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/blob/v1.7.0/observability/prometheus/rules/rabbitmq/low-disk-watermark-predicted.yml",children:"LowDiskWatermarkPredicted"})}),": Available disk space is predicted to run out within 24h. Limit queue backlogs, consume faster or increase disk size."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/blob/v1.7.0/observability/prometheus/rules/rabbitmq/file-descriptors-near-limit.yml",children:"FileDescriptorsNearLimit"})}),": 80% of available file descriptors are in use. Fewer connections, fewer durable queues, or higher file descriptor limit will help."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/blob/v1.7.0/observability/prometheus/rules/rabbitmq/tcp-sockets-near-limit.yml",children:"TCPSocketsNearLimit"})}),": 80% of available TCP sockets are in use. More channels, fewer connections or a more even spread across the cluster will help."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/blob/v1.7.0/observability/prometheus/rules/rabbitmq/container-restarts.yml",children:"ContainerRestarts"})}),": The Erlang VM system process within which RabbitMQ runs had an abnormal exit. The most common cause is misconfiguration."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/blob/v1.7.0/observability/prometheus/rules/rabbitmq-cluster-operator/unavailable-replicas.yml",children:"RabbitMQClusterOperatorUnavailableReplicas"})}),": The Operator managing RabbitMQ clusters is not available. Pod scheduling or misconfiguration issue."]}),"\n"]}),"\n",(0,a.jsx)(t.h2,{id:"how-to-get-started-quickly",children:"How to get started quickly?"}),"\n",(0,a.jsx)(t.p,{children:"You will need the following:"}),"\n",(0,a.jsxs)(t.ol,{children:["\n",(0,a.jsx)(t.li,{children:"Any Kubernetes deployment version 1.18 or above"}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.a,{href:"//kubernetes.io/docs/tasks/tools/",children:(0,a.jsx)(t.code,{children:"kubectl"})})," pointing to your Kubernetes deployment and matching the Kubernetes server version"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.a,{href:"//helm.sh/docs/intro/install/",children:(0,a.jsx)(t.code,{children:"helm"})})," version 3"]}),"\n"]}),"\n",(0,a.jsx)(t.p,{children:"Now you are ready to run the following in your terminal:"}),"\n",(0,a.jsx)(t.pre,{children:(0,a.jsx)(t.code,{className:"language-bash",children:"git clone https://github.com/rabbitmq/cluster-operator.git\n\n# Optionally, set the name of the Slack channel and the Slack Webhook URL\n# If you don't have a Slack Webhook URL, create one via https://api.slack.com/messaging/webhooks\n# export SLACK_CHANNEL='#my-channel'\n# export SLACK_API_URL='https://hooks.slack.com/services/paste/your/token'\n\n./cluster-operator/observability/quickstart.sh\n"})}),"\n",(0,a.jsx)(t.p,{children:"The last command takes about 5 minutes, and it sets up the entire RabbitMQ on Kubernetes stack:"}),"\n",(0,a.jsxs)(t.ul,{children:["\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator",children:"RabbitMQ Cluster Operator"})," declares ",(0,a.jsx)(t.code,{children:"RabbitmqCluster"})," as a custom resource definition (CRD) and manages all RabbitMQ clusters in Kubernetes"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.a,{href:"https://github.com/prometheus-community/helm-charts/tree/main/charts/kube-prometheus-stack",children:"kube-prometheus-stack"})," Helm chart which installs:","\n",(0,a.jsxs)(t.ul,{children:["\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.a,{href:"https://github.com/prometheus-operator/prometheus-operator",children:"Prometheus Operator"})," manages Prometheus and Alertmanager, adds ",(0,a.jsx)(t.code,{children:"PrometheusRule"})," and ",(0,a.jsx)(t.code,{children:"ServiceMonitor"})," custom resource definitions"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.a,{href:"https://github.com/prometheus/prometheus",children:"Prometheus"})," scrapes (i.e. reads) metrics from all RabbitMQ nodes, stores metrics in a time series database, evaluates alerting rules"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.a,{href:"https://github.com/prometheus/alertmanager",children:"Alertmanager"})," receives alerts from Prometheus, groups them by RabbitMQ cluster, optionally sends notifications to Slack (or other services)"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.a,{href:"https://github.com/grafana/grafana",children:"Grafana"})," visualises metrics from Prometheus"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.a,{href:"https://github.com/kubernetes/kube-state-metrics",children:"kube-state-metrics"})," provides Kubernetes metrics RabbitMQ alerting rules rely on"]}),"\n"]}),"\n"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/blob/v1.7.0/observability/prometheus/monitors/rabbitmq-servicemonitor.yml",children:"ServiceMonitor"})," configuration for Prometheus which helps discover ",(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/rabbitmq-server/blob/master/deps/rabbitmq_prometheus/metrics.md",children:"RabbitMQ metrics"})," from all RabbitMQ nodes"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/tree/v1.7.0/observability/prometheus/rules",children:"PrometheusRule"})," for each RabbitMQ Prometheus alert condition"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/tree/v1.7.0/observability/prometheus/alertmanager",children:"Secret"})," for the Alertmanager Slack configuration (optional)"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/tree/v1.7.0/observability/grafana/dashboards",children:"ConfigMap"})," for each RabbitMQ Grafana dashboard definition"]}),"\n"]}),"\n",(0,a.jsx)(t.h2,{id:"trigger-your-first-rabbitmq-alert",children:"Trigger your first RabbitMQ alert"}),"\n",(0,a.jsx)(t.p,{children:"To trigger an alert, we need a RabbitMQ cluster. This is the easiest way to create one:"}),"\n",(0,a.jsx)(t.pre,{children:(0,a.jsx)(t.code,{className:"language-bash",children:'# Add kubectl-rabbitmq plugin to PATH so that it can be used directly\nexport PATH="$PWD/cluster-operator/bin:$PATH"\n\n# Use kubectl-rabbitmq plugin to create RabbitmqCluster
1s via kubectl\nkubectl rabbitmq create myrabbit --replicas 3\n'})}),"\n",(0,a.jsxs)(t.p,{children:["To trigger the ",(0,a.jsx)(t.code,{children:"NoMajorityOfNodesReady"})," alert, we stop the ",(0,a.jsx)(t.code,{children:"rabbit"})," application on two out of three nodes:"]}),"\n",(0,a.jsx)(t.pre,{children:(0,a.jsx)(t.code,{className:"language-bash",children:"kubectl exec myrabbit-server-0 --container rabbitmq -- rabbitmqctl stop_app\nkubectl exec myrabbit-server-1 --container rabbitmq -- rabbitmqctl stop_app\n"})}),"\n",(0,a.jsxs)(t.p,{children:["Within 2 minutes, two out of three RabbitMQ nodes will be shown as not ",(0,a.jsx)(t.code,{children:"READY"}),":"]}),"\n",(0,a.jsx)(t.pre,{children:(0,a.jsx)(t.code,{className:"language-diff",children:"kubectl rabbitmq get myrabbit\nNAME                    READY   STATUS    RESTARTS   AGE\n- pod/myrabbit-server-0   1/1     Running   0          70s\n+ pod/myrabbit-server-0   0/1     Running   0          3m\n- pod/myrabbit-server-1   1/1     Running   0          70s\n+ pod/myrabbit-server-1   0/1     Running   0          3m\n  pod/myrabbit-server-2   1/1     Running   0          3m\n"})}),"\n",(0,a.jsxs)(t.p,{children:["The pods are still ",(0,a.jsx)(t.code,{children:"Running"}
1)," because the ",(0,a.jsx)(t.code,{children:"rabbitmqctl stop_app"})," command leaves the Erlang VM system process running."]}),"\n",(0,a.jsxs)(t.p,{children:["To see the ",(0,a.jsx)(t.code,{children:"NoMajorityOfNodesReady"})," alert triggered in Prometheus, we open the Prometheus UI in our browser: ",(0,a.jsx)(t.a,{href:"http://localhost:9090/alerts",children:"http://localhost:9090/alerts"}),".\nFor this to work, we forward local port 9090 to Prometheus port 9090 running inside Kubernetes:"]}),"\n",(0,a.jsx)(t.pre,{children:(0,a.jsx)(t.code,{className:"language-bash",children:"kubectl -n kube-prometheus port-forward svc/prom-kube-prometheus-stack-prometheus 9090\n"})}),"\n",(0,a.jsx)(t.p,{children:(0,a.jsx)(t.img,{src:r(56312).A+"",width:"2786",height:"2378"})}),"\n",(0,a.jsxs)(t.p,{children:[(0,a.jsx)(t.code,{children:"NoMajorityOfNodesReady"})," alert is first orange which means it is in a ",(0,a.jsx)(t.code,{children:"pending"})," state.\nAfter 5 minutes the colour changes to red and the state becomes ",(0,a.jsx)(t.code,{children:"firing"}),".\nThis will send an alert to Alertmanager.\nAfter we port-forward - same as above - we open the Alertmanager UI: ",(0,a.jsx)(t.a,{href:"http://localhost:9093",children:"http://localhost:9093"})]}),"\n",(0,a.jsx)(t.pre,{children:(0,a.jsx)(t.code,{className:"language-bash",children:"kubectl -n kube-prometheus port-forward svc/prom-kube-prometheus-stack-alertmanager 9093\n"})}),"\n",(0,a.jsx)(t.p,{children:(0,a.jsx)(t.img,{src:r(82857).A+"",width:"2176",height:"1480"})}),"\n",(0,a.jsxs)(t.p,{children:["Alertmanager groups alerts by ",(0,a.jsx)(t.code,{children:"namespace"})," and ",(0,a.jsx)(t.code,{children:"rabbitmq_cluster"}),".\nYou see a single alert which Alertmanager forwards to your configured Slack channel:"]}),"\n",(0,a.jsx)(t.p,{children:(0,a.jsx)(t.img,{src:r(50298).A+"",width:"839",height:"454"})}),"\n",(0,a.jsxs)(t.p,{children:["Congratulations, you triggered your first RabbitMQ alert! To resolve the alert, start the ",(0,a.jsx)(t.code,{children:"rabbit"})," application on both nodes:"]}),"\n",(0,a.jsx)(t.pre,{children:(0,a.jsx)(t.code,{className:"language-bash",children:"kubectl exec myrabbit-server-0 --container rabbitmq -- rabbitmqctl start_app\nkubectl exec myrabbit-server-1 --container rabbitmq -- rabbitmqctl start_app\n"})}),"\n",(0,a.jsxs)(t.p,{children:["The alert will transition to green in Prometheus, it will be removed from Alertmanager, and a ",(0,a.jsx)(t.strong,{children:"RESOLVED"})," notification will be sent to your Slack channel."]}),"\n",(0,a.jsx)(t.h2,{id:"past-and-current-rabbitmq-alerts",children:"Past and current RabbitMQ alerts"}),"\n",(0,a.jsxs)(t.p,{children:["To see all past and current RabbitMQ alerts across all your RabbitMQ clusters, look at the RabbitMQ-Alerts Grafana dashboard: ",(0,a.jsx)(t.a,{href:"http://localhost:3000/d/jjCq5SLMk",children:"http://localhost:3000/d/jjCq5SLMk"})," (username: ",(0,a.jsx)(t.code,{children:"admin"})," & password: ",(0,a.jsx)(t.code,{children:"admin"}),")"]}),"\n",(0,a.jsx)(t.pre,{children:(0,a.jsx)(t.code,{className:"language-bash",children:"kubectl -n kube-prometheus port-forward svc/prom-grafana 3000:80\n"})}),"\n",(0,a.jsx)(t.p,{children:(0,a.jsx)(t.img,{src:r(10857).A+"",width:"5120",height:"2158"})}),"\n",(0,a.jsx)(t.p,{children:"In the example above, we have triggered multiple alerts across multiple RabbitMQ clusters."}),"\n",(0,a.jsx)(t.h2,{id:"how-can-you-help",children:"How can you help?"}),"\n",(0,a.jsxs)(t.p,{children:["We have shared the simplest and most useful alerts that we could think of.\nSome of you already asked us about missing alerts such as memory threshold, Erlang processes & atoms, message redeliveries etc.\n",(0,a.jsx)(t.a,{href:"https://tanzu.vmware.com/rabbitmq",children:"Commercial customers"})," asked us for runbooks and automated alert resolution."]}),"\n",(0,a.jsxs)(t.p,{children:["What are your thoughts on the current alerting rules? What alerts are you missing?\n",(0,a.jsx)(t.a,{href:"https://github.com/rabbitmq/cluster-operator/discussions",children:"Let us know via a GitHub discussion."})]})]})}function b(e={}){let{wrapper:t}={...(0,i.R)(),...e.components};return t?(0,a.jsx)(t,{...e,children:(0,a.jsx)(c,{...e})}):c(e)}},82857(e,t,r){r.d(t,{A:()=>s});let s=r.p+"assets/images/alertmanager-a37760a3db7450e7c0d8876a4198ca89.png"},56312(e,t,r){r.d(t,{A:()=>s});let s=r.p+"assets/images/prometheus-alerts-6e8381b5b92bcaf7446ba5
1da9d5bcf26.png"},10857(e,t,r){r.d(t,{A:()=>s});let s=r.p+"assets/images/rabbitmq-alerts-dashboard-40330f79f0fde0b50eb3e0b0b4e083d8.png"},50298(e,t,r){r.d(t,{A:()=>s});let s=r.p+"assets/images/slack-no-majority-of-nodes-ready-e225b2376a5ff5eaeec0da3033091ecd.png"},92631(e,t,r){r.d(t,{A:()=>s});let s=r.p+"assets/images/slack-unroutable-messages-c588a5333fd4e3f2eb86549e00081f56.png"},28453(e,t,r){r.d(t,{R:()=>n,x:()=>o});var s=r(96540);let a={},i=s.createContext(a);function n(e){let t=s.useContext(i);return s.useMemo(function(){return"function"==typeof e?e(t):{...t,...e}},[t,e])}function o(e){let t;return t=e.disableParentContext?"function"==typeof e.components?e.components(a):e.components||a:n(e.components),s.createElement(i.Provider,{value:t},e.children)}},94308(e){e.exports=JSON.parse('{"permalink":"/blog/2021/05/03/alerting","editUrl":"https://github.com/rabbitmq/rabbitmq-website/tree/main/blog/2021-05-03-alerting/index.md","source":"@site/blog/2021-05-03-alerting/index.md","title":"Notify me when RabbitMQ has a problem","description":"If you want to be notified when your RabbitMQ deployments have a problem, now you can set up the RabbitMQ monitoring and alerting that we have made available in the RabbitMQ Cluster Operator repository.","date":"2021-05-03T00:00:00.000Z","tags":[{"inline":true,"label":"Kubernetes","permalink":"/blog/tags/kubernetes"},{"inline":true,"label":"New Features","permalink":"/blog/tags/new-features"}],"readingTime":7,"hasTruncateMarker":true,"authors":[{"name":"David Ansari","url":"https://github.com/ansd","socials":{"github":"https://github.com/ansd","linkedin":"https://www.linkedin.com/in/ansd/","mastodon":"https://m.ansd.xyz/@ansd","bluesky":"https://bsky.app/profile/ansd.xyz"},"imageURL":"https://github.com/ansd.png","key":"dansari","page":null},{"name":"Gerhard Lazu","key":"glazu","page":null}],"frontMatter":{"title":"Notify me when RabbitMQ has a problem","tags":["Kubernetes","New Features"],"authors":["dansari","glazu"]},"unlisted":false,"prevItem":{"title":"RabbitMQ 3.9.0 release calendar","permalink":"/blog/2021/07/09/rabbitmq-3.9.0-release-calendar"},"nextItem":{"title":"Preparing for the Bintray Shutdown: How to Migrate","permalink":"/blog/2021/03/31/migrate-off-of-bintray"}}')}}]);

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.