1"use strict";(self.webpackChunkrabbitmq_website=self.webpackChunkrabbitmq_website||[]).push([["11421"],{74533(e,n,s){s.r(n),s.d(n,{metadata:()=>i,default:()=>h,frontMatter:()=>o,contentTitle:()=>a,toc:()=>l,assets:()=>c});var i=JSON.parse('{"id":"cluster-formation","title":"Cluster Formation and Peer Discovery","description":"\x3c!--","source":"@site/versioned_docs/version-4.0/cluster-formation.md","sourceDirName":".","slug":"/cluster-formation","permalink":"/docs/4.0/cluster-formation","draft":false,"unlisted":false,"editUrl":"https://github.com/rabbitmq/rabbitmq-website/tree/main/versioned_docs/version-4.0/cluster-formation.md","tags":[],"version":"4.0","frontMatter":{"title":"Cluster Formation and Peer Discovery"},"sidebar":"docsSidebar","previous":{"title":"Clustering Guide","permalink":"/docs/4.0/clustering"},"next":{"title":"Network Partitions","permalink":"/docs/4.0/partitions"}}'),r=s(74848),t=s(28453);let o={title:"Cluster Formation and Peer Discovery"},a="Cluster Formation and Peer Discovery",c={},l=[{value:"Overview",id:"overview",level:2},{value:"What is Peer Discovery?",id:"peer-discovery",level:2},{value:"Available Discovery Mechanisms",id:"peer-discovery-plugins",level:3},{value:"Specifying the Peer Discovery Mechanism",id:"peer-discovery-configuring-mechanism",level:3},{value:"How Peer Discovery Works",id:"peer-discovery-how-does-it-work",level:3},{value:"Cluster Formation and Feature Availability",id:"formation-and-availability",level:2},{value:"Nodes Rejoining Their Existing Cluster",id:"rejoining",level:2},{value:"How to Configure Peer Discovery",id:"configuring",level:2},{value:"Config File Peer Discovery Backend",id:"peer-discovery-classic-config",level:2},{value:"Config File Peer Discovery Overview",id:"config-file-peer-discovery-overview",level:3},{value:"Configuration",id:"configuration",level:3},{value:"DNS Peer Discovery Backend",id:"peer-discovery-dns",level:2},{value:"DNS Peer Discovery Overview",id:"dns-peer-discovery-overview",level:3},{value:"Configuration",id:"configuration-1",level:3},{value:"DNS Peer Discovery and the Parallel Cluster Formation Race Condition",id:"dns-peer-discovery-and-the-parallel-cluster-formation-race-condition",level:3},{value:"Host File Modifications in Containerized Environments",id:"host-file-modifications-in-containerized-environments",level:3},{value:"Peer Discovery on AWS (EC2)",id:"peer-discovery-aws",level:2},{value:"AWS Peer Discovery Overview",id:"aws-peer-discovery-overview",level:3},{value:"Configuration and Credentials",id:"peer-discovery-aws-credentials",level:3},{value:"Using Autoscaling Group Membership",id:"peer-discovery-aws-autoscaling-group-membership",level:3},{value:"Using EC2 Instance Tags",id:"peer-discovery-aws-tags",level:3},{value:"Using Private EC2 Instance IPs",id:"peer-discovery-aws-other-settings",level:3},{value:"Peer Discovery on Kubernetes",id:"peer-discovery-k8s",level:2},{value:"Kubernetes Peer Discovery Overview",id:"kubernetes-peer-discovery-overview",level:3},{value:"Important: Prerequisites and Deployment Considerations",id:"important-prerequisites-and-deployment-considerations",level:3},{value:"Use a Stateful Set",id:"use-a-stateful-set",level:4},{value:"Use Persistent Volumes",id:"use-persistent-volumes",level:4},{value:"Make Sure <code>/etc/rabbitmq</code> is Mounted as Writeable",id:"make-sure-etcrabbitmq-is-mounted-as-writeable",level:4},{value:"Use Parallel podManagementPolicy",id:"use-parallel-podmanagementpolicy",level:4},{value:"Use Most Basic Health Checks for RabbitMQ Pod Readiness Probes",id:"use-most-basic-health-checks-for-rabbitmq-pod-readiness-probes",level:4},{value:"Examples",id:"examples",level:3},{value:"Configuration",id:"configuration-2",level:3},{value:"Kubernetes API Endpoint",id:"kubernetes-api-endpoint",level:4},{value:"Kubernetes API Access Token",id:"kubernetes-api-access-token",level:4},{value:"Kubernetes Namespace",id:"kubernetes-namespace",level:4},{value:"Kubernetes API CA Certificate Bundle",id:"kubernetes-api-ca-certificate-bundle",level:4},{value:"Peer Node Pods Can Use Hostnames or IP Addresses",id:"peer-node-pods-can-use-hostnames-or-ip-addresses",level:4},{value:"Peer Node Pod Name Suffix",id:"peer-node-pod-name-suffix",level:5},{value:"Peer Discovery Using Consul",id:"peer-discovery-consul",level:2},{value:"Consul Peer Discovery Overview",id:"consul-peer-discovery-overview",level:3},{value:"Configuration",id:"configuration-3",level:3},{value:"Consul Endpoint",id:"consul-endpoint",level:4},{value:"Consul ACL Token",id:"consul-acl-token",level:4},{value:"Service Address",id:"service-address",level:4},{value:"Service Port",id:"service-port",level:4},{value:"Service Tags and Metadata",id:"service-tags-and-metadata",level:4},{value:"Service Health Checks",id:"service-health-checks",level:4},{value:"Opting Out of Regisration",id:"opting-out-of-regisration",level:4},{value:"Node Name Suffixes",id:"node-name-suffixes",level:4},{value:"Distributed Lock Acquisition",id:"distributed-lock-acquisition",level:4},{value:"Peer Discovery Using Etcd",id:"peer-discovery-etcd",level:2},{value:"Etcd Peer Discovery Overview",id:"etcd-peer-discovery-overview",level:3},{value:"Configuration",id:"configuration-4",level:3},{value:"etcd Endpoints and Authentication",id:"etcd-endpoints-and-authentication",level:4},{value:"Key Naming",id:"key-naming",level:4},{value:"Key Leases and TTL",id:"key-leases-and-ttl",level:4},{value:"Locks",id:"locks",level:4},{value:"Inspecting Keys",id:"inspecting-keys",level:4},{value:"TLS",id:"tls",level:4},{value:"Race Conditions During Initial Cluster Formation",id:"initial-formation-race-condition",level:2},{value:"Node Health Checks and Forced Removal",id:"node-health-checks-and-cleanup",level:2},{value:"Negative Side Effects of Automatic Removal",id:"negative-side-effects-of-automatic-removal",level:3},{value:"Peer Discovery Failures and Retries",id:"discovery-retries",level:2},{value:"HTTP Proxy Settings",id:"http-proxy-settings",level:2},{value:"Troubleshooting",id:"troubleshooting",level:2}];function d(e){let n={a:"a",admonition:"admonition",code:"code",h1:"h1",h2:"h2",h3:"h3",h4:"h4",h
15:"h5",header:"header",li:"li",ol:"ol",p:"p",pre:"pre",strong:"strong",ul:"ul",...(0,t.R)(),...e.components};return(0,r.jsxs)(r.Fragment,{children:[(0,r.jsx)(n.header,{children:(0,r.jsx)(n.h1,{id:"cluster-formation-and-peer-discovery",children:"Cluster Formation and Peer Discovery"})}),"\n",(0,r.jsx)(n.h2,{id:"overview",children:"Overview"}),"\n",(0,r.jsxs)(n.p,{children:["This guide covers various automation-oriented cluster formation and\npeer discovery features. For a general overview of RabbitMQ clustering,\nplease refer to the ",(0,r.jsx)(n.a,{href:"./clustering",children:"Clustering Guide"}),"."]}),"\n",(0,r.jsxs)(n.p,{children:["This guide assumes general familiarity with ",(0,r.jsx)(n.a,{href:"./clustering",children:"RabbitMQ clustering"}),"\nand focuses on the peer discovery subsystem.\nFor example, it will not cover what ",(0,r.jsx)(n.a,{href:"./networking",children:"ports must be open"})," for inter-node communication, how nodes authenticate to each other, and so on.\nBesides discovery mechanisms and ",(0,r.jsx)(n.a,{href:"#configuring",children:"their configuration"}),",\nthis guide also covers closely related topics of ",(0,r.jsx)(n.a,{href:"#formation-and-availability",children:"feature availability during cluster formation"}),", ",(0,r.jsx)(n.a,{href:"#rejoining",children:"rejoining nodes"}),",\nthe problem of ",(0,r.jsx)(n.a,{href:"#initial-formation-race-condition",children:"initial cluster formation"})," with nodes booting in parallel as well as ",(0,r.jsx)(n.a,{href:"#node-health-checks-and-cleanup",children:"additional health checks"})," offered\nby some discovery implementations."]}),"\n",(0,r.jsxs)(n.p,{children:["The guide also covers the basics of ",(0,r.jsx)(n.a,{href:"#troubleshooting",children:"peer discovery troubleshooting"}),"."]}),"\n",(0,r.jsx)(n.h2,{id:"peer-discovery",children:"What is Peer Discovery?"}),"\n",(0,r.jsx)(n.p,{children:'To form a cluster, new ("blank") nodes need to be able to discover\ntheir peers. This can be done using a variety of mechanisms (backends).\nSome mechanisms assume all cluster members are known ahead of time (for example, listed\nin the config file), others are dynamic (nodes can come and go).'}),"\n",(0,r.jsx)(n.p,{children:"All peer discovery mechanisms assume that newly joining nodes will be able to\ncontact their peers in the cluster and authenticate with them successfully.\nThe mechanisms that rely on an external service (e.g. DNS or Consul) or API (e.g. AWS or Kubernetes)\nrequire the service(s) or API(s) to be available and reachable on their standard ports.\nInability to reach the services will lead to node's inability to join the cluster."}),"\n",(0,r.jsx)(n.h3,{id:"peer-discovery-plugins",children:"Available Discovery Mechanisms"}),"\n",(0,r.jsx)(n.p,{children:"The following mechanisms are built into the core and always available:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsx)(n.li,{children:(0,r.jsx)(n.a,{href:"#peer-discovery-classic-config",children:"Config file"})}),"\n",(0,r.jsx)(n.li,{children:(0,r.jsx)(n.a,{href:"#peer-discovery-dns",children:"Pre-configured DNS A/AAAA records"})}),"\n"]}),"\n",(0,r.jsxs)(n.p,{children:["Additional peer discovery mechanisms are available via plugins. The following\npeer discovery plugins ship with ",(0,r.jsx)(n.a,{href:"/release-information",children:"supported RabbitMQ versions"}),":"]}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsx)(n.li,{children:(0,r.jsx)(n.a,{href:"#peer-discovery-aws",children:"AWS (EC2)"})}),"\n",(0,r.jsx)(n.li,{children:(0,r.jsx)(n.a,{href:"#peer-discovery-k8s",children:"Kubernetes"})}),"\n",(0,r.jsx)(n.li,{children:(0,r.jsx)(n.a,{href:"#peer-discovery-consul",children:"Consul"})}),"\n",(0,r.jsx)(n.li,{children:(0,r.jsx)(n.a,{href:"#peer-discovery-etcd",children:"etcd"})}),"\n"]}),"\n",(0,r.jsxs)(n.p,{children:["The above plugins do not need to be installed but like all ",(0,r.jsx)(n.a,{href:"./plugins",children:"plugins"})," they must be ",(0,r.jsx)(n.a,{href:"./plugins#basics",children:"enabled"}),"\nor ",(0,r.jsx)(n.a,{href:"./plugins#enabled-plugins-file",children:"preconfigured"})," before they can be used."]}),"\n",(0,r.jsxs)(n.p,{children:["For peer discovery plugins, which must be available on node boot, this means they must be enabled before first node boot.\nThe example below uses ",(0,r.jsx)(n.a,{href:"./cli",children:"rabbitmq-plugins"}),"' ",(0,r.jsx)(n.code,{children:"--offline"})," mode:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-bash",children:"rabbitmq-plugins --offline enable <plugin name>\n"})}),"\n",(0,r.jsx)(n.p,{children:"A more specific example:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-bash",children:"rabbitmq-plugins --offline enable rabbitmq_peer_discovery_k8s\n"})}),"\n",(0,r.jsx)(n.p,{children:"A node with configuration settings that belong a non-enabled peer discovery plugin will fail\nto start and report those settings as unknown."}),"\n",(0,r.jsx)(n.h3,{id:"peer-discovery-configuring-mechanism",children:"Specifying the Peer Discovery Mechanism"}),"\n",(0,r.jsxs)(n.p,{children:["The discovery mechanism to use is specified in the ",(0,r.jsx)(n.a,{href:"./configure",children:"config file"}),",\nas are various mechanism-specific settings, for example, discovery service hostnames, credentials, and so\non. ",(0,r.jsx)(n.code,{children:"cluster_formation.peer_discovery_backend"})," is the key\nthat controls what discovery module (implementation) is used:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = classic_config\n\n# The backend can also be specified using its module name. Note that\n# module names do not necessarily match plugin names exactly.\n# cluster_formation.peer_discovery_backend = rabbit_peer_discovery_classic_config\n"})}),"\n",(0,r.jsxs)(n.p,{children:["The module has to implement the ",(0,r.jsx)(n.a,{href:"https://github.com/rabbitmq/rabbitmq-common/blob/master/src/rabbit_peer_discovery_backend.erl",children:"rabbit_peer_discovery_backend"}),"\nbehaviour. Plugins therefore can introduce their own discovery\nmechanisms."]}),"\n",(0,r.jsx)(n.h3,{id:"peer-discovery-how-does-it-work",children:"How Peer Discovery Works"}),"\n",(0,r.jsx)(n.p,{children:"When a node starts and detects it doesn't have a previously\ninitialised database, it will check if there's a peer\ndiscovery mechanism configured. If that's the case, it will\nthen perform the discovery and attempt to contact each\ndiscovered peer in order. Finally, it will attempt to join the\ncluster of the first reachable peer."}),"\n",(0,r.jsx)(n.p,{children:"Depending on the backend (mechanism) used, the process of peer discovery may involve\ncontacting external services, for example, an AWS API endpoint, a Consul node or\nperforming a DNS query. Some backends require nodes to register (tell the backend that the\nnode is up and should be counted as a cluster member): for example, Consul and etcd both\nsupport registration. With other backends the list of nodes is configured ahead of\ntime (e.g. config file). Those backends are said to not support node registration."}),"\n",(0,r.jsxs)(n.p,{children:["In some cases node registration is implicit or managed by an external service.\nAWS autoscaling groups is a good example: AWS keeps track of group membership,\nso nodes don't have to (or cannot) explicitly register. However, the list of cluster members\nis not predefined. Such backends usually include a no-op registration step\nand apply one of the ",(0,r.jsx)(n.a,{href:"#initial-formation-race-condition",children:"race condition mitigation mechanisms"})," described below."]}),"\n",(0,r.jsx)(n.p,{children:"If the configured backend supports registration, nodes unregister when they are instructed to stop."}),"\n",(0,r.jsxs)(n.p,{children:["It is possible to opt-out of registration completely with the config option ",(0,r.jsx)(n.code,{children:"cluster_formation.registration.enabled"}),":"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.registration.enabled = false\n"})}),"\n",(0,r.jsxs)(n.p,{children:["When configured this way, the node has to be registered manually or using another mechanism,\ne.g. by a container orchestrator such as ",(0,r.jsx)(n.a,{href:"https://developer.hashicorp.com/nomad/integrations/hashicorp/rabbitmq",children:"Nomad"})," or ",(0,r.jsx)(n.a,{href:"https://www.rabbitmq.com/kubernetes/operator/operator-overview",children:"Kubernetes"}),"."]}),"\n",(0,r.jsxs)(n.p,{children:["If peer discovery isn't configured, or it ",(0,r.jsx)(n.a,{href:"#discovery-retries",children:"repeatedly fails"}),",\nor no peers are reachable, a node that wasn't a cluster member in the past\nwill initialise from scratch and proceed as a standalone node.\nPeer discovery progress and outcomes will be ",(0,r.jsx)(n.a,{href:"./logging",children:"logged"}),"\nby the node."]}),"\n",(0,r.jsx)(n.p,{children:'If a node previously was a cluster member, it will try to contact and rejoin\nits "last seen" peer for a period of time. In this case, no peer discovery\nwill be performed. This is true for all backends.'}),"\n",(0,r.jsx)(n.h2,{id:"formation-and-availability",children:"Cluster Formation and Feature Availability"}),"\n",(0,r.jsxs)(n.p,{children:["As a general rule, a cluster that is only been partly formed, that is, only a subset of\nnodes has joined it ",(0,r.jsx)(n.strong,{children:"must be considered fully available"})," by clients."]}),"\n",(0,r.jsxs)(n.p,{children:["Individual nodes will accept ",(0,r.jsx)(n.a,{href:"./connections",children:"client connections"})," before the cluster is formed. In such cases,\nclients should be prepared to certain features not being available. For instance, ",(0,r.jsx)(n.a,{href:"./quorum-queues",children:"quorum queues"}),"\nwon't be available unless the number of cluster nodes matches or exceeds the quorum of configured replica count."]}),"\n",(0,r.jsxs)(n.p,{children:["Features behind ",(0,r.jsx)(n.a,{href:"./feature-flags",children:"feature flags"})," may also be unavailable until cluster formation completes."]}),"\n",(0,r.jsx)(n.h2,{id:"rejoining",children:"Nodes Rejoining Their Existing Cluster"}),"\n",(0,r.jsx)(n.p,{children:"A new node joining a cluster is just one possible case. Another common sce
1nario\nis when an existing cluster member temporarily leaves and then rejoins the cluster.\nWhile the peer discovery subsystem does not affect the behavior described in this section,\nit's important to understand how nodes behave when they rejoin their cluster after a restart or failure."}),"\n",(0,r.jsxs)(n.p,{children:["Existing cluster members ",(0,r.jsx)("strong",{children:"will not perform peer discovery"}),". Instead they will try to\ncontact their previously known peers."]}),"\n",(0,r.jsx)(n.p,{children:'If a node previously was a cluster member, when it boots it will try to contact\nits "last seen" peer for a period of time. If the peer is not booted (e.g. when\na full cluster restart or upgrade is performed) or cannot be reached, the node will\nretry the operation a number of times.'}),"\n",(0,r.jsxs)(n.p,{children:["Default values are ",(0,r.jsx)(n.code,{children:"10"})," retries and ",(0,r.jsx)(n.code,{children:"30"})," seconds per attempt,\nrespectively, or 5 minutes total. In environments where nodes can take a long and/or uneven\ntime to start it is recommended that the number of retries is increased."]}),"\n",(0,r.jsxs)(n.p,{children:["If a node is reset since losing contact with the cluster, it will behave ",(0,r.jsx)(n.a,{href:"#peer-discovery-how-does-it-work",children:"like a blank node"}),".\nNote that other cluster members might still consider it to be a cluster member, in which case\nthe two sides will disagree and the node will fail to join. Such reset nodes must also be\nremoved from the cluster using ",(0,r.jsx)(n.a,{href:"./cli",children:(0,r.jsx)(n.code,{children:"rabbitmqctl forget_cluster_node"})})," executed against\nan existing cluster member."]}),"\n",(0,r.jsxs)(n.p,{children:["If a node was explicitly removed from the cluster by the operator and then reset,\nit will be able to join the cluster as a new member. In this case it will behave exactly\n",(0,r.jsx)(n.a,{href:"#peer-discovery-how-does-it-work",children:"like a blank node"})," would."]}),"\n",(0,r.jsxs)(n.p,{children:["A node rejoining after a node name or host name change can start as ",(0,r.jsx)(n.a,{href:"#peer-discovery-how-does-it-work",children:"a blank node"}),"\nif its data directory path changes as a result. Such nodes will fail to rejoin the cluster.\nWhile the node is offline, its peers can be reset or started with a blank data directory.\nIn that case the recovering node will fail to rejoin its peer as well since internal data store cluster\nidentity would no longer match."]}),"\n",(0,r.jsx)(n.p,{children:"Consider the following scenario:"}),"\n",(0,r.jsxs)("ol",{children:[(0,r.jsx)("li",{children:"A cluster of 3 nodes, A, B and C is formed"}),(0,r.jsx)("li",{children:"Node A is shut down"}),(0,r.jsx)("li",{children:"Node B is reset"}),(0,r.jsx)("li",{children:"Node A is started"}),(0,r.jsx)("li",{children:"Node A tries to rejoin B but B's cluster identity has changed"}),(0,r.jsx)("li",{children:"Node B doesn't recognise A as a known cluster member because it's been reset"})]}),"\n",(0,r.jsx)(n.p,{children:"in this case node B will reject the clustering attempt from A with an appropriate error\nmessage in the log:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{children:"Node '[email protected]' thinks it's clustered with node '[email protected]', but '[email protected]' disagrees\n"})}),"\n",(0,r.jsx)(n.p,{children:"In this case B can be reset again and then will be able to join A, or A\ncan be reset and will successfully join B."}),"\n",(0,r.jsx)(n.h2,{id:"configuring",children:"How to Configure Peer Discovery"}),"\n",(0,r.jsxs)(n.p,{children:["Peer discovery plugins are configured just like the core server and other\nplugins: using a ",(0,r.jsx)(n.a,{href:"./configure",children:"config file"}),"."]}),"\n",(0,r.jsxs)(n.p,{children:[(0,r.jsx)(n.code,{children:"cluster_formation.peer_discovery_backend"})," is the key that ",(0,r.jsx)(n.a,{href:"#peer-discovery-configuring-mechanism",children:"controls what peer discovery backend will be used"}),".\nEach backend will also have a number of configuration settings specific to it.\nThe rest of the guide will c
1over configurable settings specific to a particular mechanism\nas well as provide examples for each one."]}),"\n",(0,r.jsx)(n.h2,{id:"peer-discovery-classic-config",children:"Config File Peer Discovery Backend"}),"\n",(0,r.jsx)(n.h3,{id:"config-file-peer-discovery-overview",children:"Config File Peer Discovery Overview"}),"\n",(0,r.jsx)(n.p,{children:"The most basic way for a node to discover its cluster peers is to read a list\nof nodes from the config file. The set of cluster members is assumed to be known at deployment\ntime."}),"\n",(0,r.jsx)(n.h3,{id:"configuration",children:"Configuration"}),"\n",(0,r.jsxs)(n.p,{children:["The peer nodes are listed using the ",(0,r.jsx)(n.code,{children:"cluster_formation.classic_config.nodes"})," config setting:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = classic_config\n\n# the backend can also be specified using its module name\n# cluster_formation.peer_discovery_backend = rabbit_peer_discovery_classic_config\n\ncluster_formation.classic_config.nodes.1 = [email protected]\ncluster_formation.classic_config.nodes.2 = [email protected]\n"})}),"\n",(0,r.jsx)(n.h2,{id:"peer-discovery-dns",children:"DNS Peer Discovery Backend"}),"\n",(0,r.jsxs)(n.admonition,{type:"important",children:[(0,r.jsxs)(n.p,{children:["This peer discovery mechanism is sensitive to OS and RabbitMQ configuration that\n",(0,r.jsx)(n.a,{href:"./networking#dns",children:"affects hostname resolution"}),"."]}),(0,r.jsxs)(n.p,{children:["For example, a deployment tool that modifies the ",(0,r.jsx)(n.a,{href:"https://en.wikipedia.org/wiki/Hosts_(file)",children:"local host file"}),"\ncan affect (break) this peer discovery mechanism."]})]}),"\n",(0,r.jsx)(n.h3,{id:"dns-peer-discovery-overview",children:"DNS Peer Discovery Overview"}),"\n",(0,r.jsx)(n.p,{children:'Another built-in peer discovery mechanism as of RabbitMQ 3.7.0 is DNS-based.\nIt relies on a pre-configured hostname ("seed hostname") with DNS A (or AAAA) records and reverse DNS lookups\nto perform peer discovery. More specifically, this mechanism will perform the following steps:'}),"\n",(0,r.jsxs)(n.ol,{children:["\n",(0,r.jsx)(n.li,{children:"Query DNS A records of the seed hostname."}),"\n",(0,r.jsx)(n.li,{children:"For each returned DNS record's IP address, perform a reverse DNS lookup."}),"\n",(0,r.jsxs)(n.li,{children:["Append current node's prefix (e.g. ",(0,r.jsx)(n.code,{children:"rabbit"})," in ",(0,r.jsx)(n.code,{children:"[email protected]"}),")\nto each hostname and return the result."]}),"\n"]}),"\n",(0,r.jsxs)(n.p,{children:["For example, let's consider a seed hostname of\n",(0,r.jsx)(n.code,{children:"discovery.eng.example.local"}),". It has 2 DNS A\nrecords that return two IP addresses:\n",(0,r.jsx)(n.code,{children:"192.168.100.1"})," and\n",(0,r.jsx)(n.code,{children:"192.168.100.2"}),". Reverse DNS lookups for those IP\naddresses return ",(0,r.jsx)(n.code,{children:"node1.eng.example.local"})," and\n",(0,r.jsx)(n.code,{children:"node2.eng.example.local"}),", respectively. Current node's name is not\nset and defaults to ",(0,r.jsx)(n.code,{children:"rabbit@$(hostname)"}),".\nThe final list of nodes discovered will contain two nodes: ",(0,r.jsx)(n.code,{children:"[email protected]"}),"\nand ",(0,r.jsx)(n.code,{children:"[email protected]"}),"."]}),"\n",(0,r.jsx)(n.h3,{id:"configuration-1",children:"Configuration"}),"\n",(0,r.jsxs)(n.p,{children:["The seed hostname is set using the ",(0,r.jsx)(n.code,{children:"cluster_formation.dns.hostname"})," config setting:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = dns\n\n# the backend can also be specified using its module name\n# cluster_formation.peer_discovery_backend = rabbit_peer_discovery_dns\n\ncluster_formation.dns.hostname = discovery.eng.example.local\n"})}),"\n",(0,r.jsx)(n.h3,{id:"dns-peer-discovery-and-the-parallel-cluster-formation-race-condition",children:"DNS Peer Discovery and the Parallel Cluster Formation Race Condition"}),"\n",(0,r.jsxs)(n.p,{children:["The DNS peer discovery backend does not use locking to prevent the ",(0,r.jsx)(n.a,{href:"#initial-formation-race-condition",children:"inherent race condition"})," during parallel initial cluster formation and requires a random startup delay to be injected by the deployment tooling."]}),"\n",(0,r.jsx)(n.p,{children:"A random value in the 1 to 20 second range is recommended\nfor most environments. In the environments where nodes are dynamically registered\nwith the DNS system and it can take time, use a broader range, such as 1 to 60 seconds."}),"\n",(0,r.jsx)(n.h3,{id:"host-file-modifications-in-containerized-environments",children:"Host File Modifications in Containerized Environments"}),"\n",(0,r.jsx)(n.admonition,{type:"warning",children:(0,r.jsxs)(n.p,{children:["In some containerized environments, the ",(0,r.jsx)(n.a,{href:"https://en.wikipedia.org/wiki/Hosts_(file)",children:"local host file"})," is modified at container\nstartup time. This can affect hostname resolution on the host and make it impossible for this peer\ndiscovery mechanism to do its job."]})}),"\n",(0,r.jsxs)(n.p,{children:["In some containerized environments, the ",(0,r.jsx)(n.a,{href:"https://en.wikipedia.org/wiki/Hosts_(file)",children:"local host file"})," is modified at container\nstartup time, for example, a configuration- or c
1onvention-based local hostname can be added to it."]}),"\n",(0,r.jsx)(n.p,{children:"This can affect hostname resolution on the host and make it impossible for this peer\ndiscovery mechanism to do its job."}),"\n",(0,r.jsxs)(n.p,{children:["Podman is one known example of a tool that can perform such host file modifications.\nIn order to avoid this, set its ",(0,r.jsxs)(n.a,{href:"https://github.com/containers/common/blob/main/docs/containers.conf.5.md",children:[(0,r.jsx)(n.code,{children:"host_containers_internal_ip"})," setting"]}),"\nmust be set to a blank string."]}),"\n",(0,r.jsxs)(n.p,{children:["In environments where container-level settings cannot be tuned, the runtime\ncan be ",(0,r.jsx)(n.a,{href:"./networking#the-inetrc-file",children:"configured to ignore the standard local hosts file"}),"\nand only use DNS or a pre-configured set of hostname-to-IP address mappings."]}),"\n",(0,r.jsx)(n.h2,{id:"peer-discovery-aws",children:"Peer Discovery on AWS (EC2)"}),"\n",(0,r.jsx)(n.h3,{id:"aws-peer-discovery-overview",children:"AWS Peer Discovery Overview"}),"\n",(0,r.jsxs)(n.p,{children:["An ",(0,r.jsx)(n.a,{href:"https://github.com/rabbitmq/rabbitmq-server/tree/main/deps/rabbitmq_peer_discovery_aws",children:"AWS (EC2)-specific"})," discovery mechanism\nis available via a plugin."]}),"\n",(0,r.jsxs)(n.p,{children:["As with any ",(0,r.jsx)(n.a,{href:"./plugins",children:"plugin"}),", it must be enabled before it\ncan be used. For peer discovery plugins it means they must be ",(0,r.jsx)(n.a,{href:"./plugins#basics",children:"enabled"}),"\nor ",(0,r.jsx)(n.a,{href:"./plugins#enabled-plugins-file",children:"preconfigured"})," before first node boot:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-bash",children:"rabbitmq-plugins --offline enable rabbitmq_peer_discovery_aws\n"})}),"\n",(0,r.jsx)(n.p,{children:"The plugin provides two ways for a node to discover its peers:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsx)(n.li,{children:"Using EC2 instance tags"}),"\n",(0,r.jsx)(n.li,{children:"Using AWS autoscaling group membership"}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:"Both methods rely on AWS-specific APIs (endpoints) and features and thus cannot work in\nother IaaS environments. Once a list of cluster member instances is retrieved,\nfinal node names are computed using instance hostnames or IP addresses."}),"\n",(0,r.jsx)(n.h3,{id:"peer-discovery-aws-credentials",children:"Configuration and Credentials"}),"\n",(0,r.jsx)(n.p,{children:"Before a node can perform any operations on AWS, it needs to have a set of\nAWS account credentials configured. This can be done in a couple of ways:"}),"\n",(0,r.jsxs)(n.ol,{children:["\n",(0,r.jsxs)(n.li,{children:["Via ",(0,r.jsx)(n.a,{href:"./configure",children:"config file"})]}),"\n",(0,r.jsxs)(n.li,{children:["Using environment variables ",(0,r.jsx)(n.code,{children:"AWS_ACCESS_KEY_ID"})," and ",(0,r.jsx)(n.code,{children:"AWS_SECRET_ACCESS_KEY"})]}),"\n"]}),"\n",(0,r.jsxs)(n.p,{children:[(0,r.jsx)(n.a,{href:"http://docs.aws.amazon.com/AWSEC2/latest/UserGuide/ec2-instance-metadata.html",children:"EC2 Instance Metadata service"})," for the region will also be consulted."]}),"\n",(0,r.jsx)(n.p,{children:"The following example snippet configures RabbitMQ to use the AWS peer discovery\nbackend and provides information about AWS region as well as a set of credentials:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = aws\n\n# the backend can also be specified using its module name\n# cluster_formation.peer_discovery_backend = rabbit_peer_discovery_aws\n\ncluster_formation.aws.region = us-east-1\ncluster_formation.aws.access_key_id = ANIDEXAMPLE\ncluster_formation.aws.secret_key = WjalrxuTnFEMI/K7MDENG+bPxRfiCYEXAMPLEKEY\n"})}),"\n",(0,r.jsxs)(n.p,{children:["If region is left unconfigured, ",(0,r.jsx)(n.code,{children:"us-east-1"})," will be used by default.\nSensitive values in configuration file can optionally ",(0,r.jsx)(n.a,{href:"./configure#configuration-encryption",children:"be encrypted"}),"."]}),"\n",(0,r.jsxs)(n.p,{children:["If an ",(0,r.jsx)(n.a,{href:"http://docs.aws.amazon.com/AWSEC2/latest/UserGuide/iam-roles-for-amazon-./ec2",children:"IAM role is assigned to EC2 instances"})," running RabbitMQ nodes,\na policy has to be used to ",(0,r.jsx)(n.a,{href:"https://docs.aws.amazon.com/AWSEC2/latest/APIReference/API_DescribeInstances.html",children:"allow said instances use EC2 Instance Metadata Service"}),".\nWhen the plugin is configured to use Autoscaling group members,\na policy has to ",(0,r.jsx)(n.a,{href:"https://docs.aws.amazon.com/autoscaling/ec2/userguide/control-access-using-iam.html",children:"grant access to describe autoscaling group members"})," (instances).\nBelow is an example of a policy that covers both use cases:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-json",children:'{\n"Version": "2012-10-17",\n"Statement": [\n {\n "Effect": "Allow",\n "Action": [\n "autoscaling:DescribeAutoScalingInstances",\n "ec2:DescribeInstances"\n ],\n "Resource": [\n "*"\n ]\n }\n ]\n}\n'})}),"\n",(0,r.jsx)(n.h3,{id:"peer-discovery-aws-autoscaling-group-membership",children:"Using Autoscaling Group Membership"}),"\n",(0,r.jsx)(n.p,{children:"When autoscaling-based peer discovery is used, current node's EC2 instance autoscaling\ngroup members will be listed and used to produce the list of discovered peers."}),"\n",(0,r.jsxs)(n.p,{children:["To use autoscaling group membership, set the ",(0,r.jsx)(n.code,{children:"cluster_formation.aws.use_autoscaling_group"})," key\nto ",(0,r.jsx)(n.code,{children:"true"}),":"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = aws\n\ncluster_formation.aws.region = us-east-1\ncluster_formation.aws.access_key_id = ANIDEXAMPLE\ncluster_formation.aws.secret_key = WjalrxuTnFEMI/K7MDENG+bPxRfiCYEXAMPLEKEY\n\ncluster_formation.aws.use_autoscaling_group = true\n"})}),"\n",(0,r.jsx)(n.h3,{id:"peer-discovery-aws-tags",children:"Using EC2 Instance Tags"}),"\n",(0,r.jsx)(n.p,{children:"When tags-based peer discovery is used, the plugin will list EC2 instances\nusing EC2 API and filter them by configured instance tags. Resulting instance set\nwill be used to produce the list of discovered peers."}),"\n",(0,r.jsxs)(n.p,{children:["Tags are configured using the ",(0,r.jsx)(n.code,{children:"cluster_formation.aws.instance_tags"})," key. The example\nbelow uses three tags: ",(0,r.jsx)(n.code,{children:"region"}),", ",(0,r.jsx)(n.code,{children:"service"}),", and ",(0,r.jsx)(n.code,{children:"environment"}),"."]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = aws\n\ncluster_formation.aws.region = us-east-1\ncluster_formation.aws.access_key_id = ANIDEXAMPLE\ncluster_formation.aws.secret_key = WjalrxuTnFEMI/K7MDENG+bPxRfiCYEXAMPLEKEY\n\ncluster_formation.aws.instance_tags.region = us-east-1\ncluster_formation.aws.instance_tags.service = rabbitmq\ncluster_formation.aws.instance_tags.environment = staging\n"})}),"\n",(0,r.jsx)(n.h3,{id:"peer-discovery-aws-other-settings",children:"Using Private EC2 Instance IPs"}),"\n",(0,r.jsxs)(n.p,{children:["By default peer discovery will use private DNS hostnames to compute node names.\nThis option is most convenient and is ",(0,r.jsx)(n.strong,{children:"highly recommended"}),"."]}),"\n",(0,r.jsxs)(n.p,{children:["However, it is possible to opt into using private IPs instead by setting\nthe ",(0,r.jsx)(n.code,{children:"cluster_formation.aws.use_private_ip"})," key to ",(0,r.jsx)(n.code,{children:"true"}),". For this setup to work,\n",(0,r.jsxs)(n.a,{href:"./configure#customise-environment",children:[(0,r.jsx)(n.code,{children:"RABBITMQ_NODENAME"})," must be set"]})," to the private IP address at node\ndeployment time."]}),"\n",(0,r.jsxs)(n.p,{children:[(0,r.jsx)(n.code,{children:"RABBITMQ_USE_LONGNAME"})," also has to be set to ",(0,r.jsx)(n.code,{children:"true"})," or an IP address won't be considered a valid\npart of node name."]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = aws\n\ncluster_formation.aws.region = us-east-1\ncluster_formation.aws.access_key_id = ANIDEXAMPLE\ncluster_formation.aws.secret_key = WjalrxuTnFEMI/K7MDENG+bPxRfiCYEXAMPLEKEY\n\ncluster_formation.aws.use_autoscaling_group = true\ncluster_formation.aws.use_private_ip = true\n"})}),"\n",(0,r.jsx)(n.h2,{id:"peer-discovery-k8s",children:"Peer Discovery on Kubernetes"}),"\n",(0,r.jsx)(n.h3,{id:"kubernetes-peer-discovery-overview",children:"Kubernetes Peer Discovery Overview"}),"\n",(0,r.jsxs)(n.p,{children:["A ",(0,r.jsx)(n.a,{href:"https://kubernetes.io/",children:"Kubernetes"}),"-based discovery mechanism\nis available via ",(0,r.jsx)(n.a,{href:"https://github.com/rabbitmq/rabbitmq-server/tree/main/deps/rabbitmq_peer_discovery_k8s",children:"a plugin"}),"."]}),"\n",(0,r.jsxs)(n.p,{children:["As with any ",(0,r.jsx)(n.a,{href:"./plugins",children:"plugin"}),", it must be enabled before it\ncan be used. For peer discovery plugins it means they must be ",(0,r.jsx)(n.a,{href:"./plugins#basics",children:"enabled"}),"\nor ",(0,r.jsx)(n.a,{href:"./plugins#enabled-plugins-file",children:"preconfigured"})," before first node boot:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-bash",children:"rabbitmq-plugins --offline enable rabbitmq_peer_discovery_k8s\n"})}),"\n",(0,r.jsx)(n.h3,{id:"important-prerequisites-and-deployment-considerations",children:"Important: Prerequisites and Deployment Considerations"}),"\n",(0,r.jsxs)(n.admonition,{type:"important",children:[(0,r.jsxs)(n.p,{children:["The recommended option for deploying RabbitMQ to Kubernetes is the ",(0,r.jsx)(n.a,{href:"/kubernetes/operator/operator-overview",children:"RabbitMQ Kubernetes Cluster Operator"}),"."]}),(0,r.jsx)(n.p,{children:"It follows the recommendations listed below."})]}),"\n",(0,r.jsx)(n.p,{children:"With this mechanism, nodes fetch a list of their peers from\na Kubernetes API endpoint using a set of configured values:\na URI scheme, host, port, as well as the token and certificate paths."}),"\n",(0,r.jsxs)(n.p,{children:["If the recommended option of the ",(0,r.jsx)(n.a,{href:"/kubernetes/operator/operator-overview",children:"RabbitMQ Kubernetes Cluster Operator"})," cannot be used,\nthere are several prerequisites and deployment choices that must be taken into\naccount when deploying RabbitMQ to Kubernetes, with this peer discovery mechanism\nand in general."]}),"\n",(0,r.jsx)(n.h4,{id:"use-a-stateful-set",children:"Use a Stateful Set"}),"\n",(0,r.jsxs)(n.p,{children:["A RabbitMQ cluster deployed to Kubernetes will use a set of pods. The set must be a ",(0,r.jsx)(n.a,{href:"https://kubernetes.io/docs/tasks/run-application/run-replicated-stateful-application/#statefulset",children:"stateful set"}),".\nA ",(0,r.jsx)(n.a,{href:"https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#limitations",children:"headless service"})," must be used to\ncontrol ",(0,r.jsx)(n.a,{href:"https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/",children:"network identity of the pods"}),"\n(their hostnames), which in turn affect RabbitMQ node names.\nOn the headless service ",(0,r.jsx)(n.code,{children:"spec"}),", field ",(0,r.jsx)(n.code,{children:"publishNotReadyAddresses"})," must be set to ",(0,r.jsx)(n.code,{children:"true"}
1)," to propagate SRV DNS records for its Pods for the purpose of peer discovery."]}),"\n",(0,r.jsxs)(n.p,{children:["In addition, since RabbitMQ nodes ",(0,r.jsx)(n.a,{href:"./clustering#hostname-resolution-requirement",children:"resolve their own and peer hostnames during boot"}),",\nCoreDNS ",(0,r.jsx)(n.a,{href:"https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#stable-network-id",children:"caching timeout may need to be decreased"})," from default 30 seconds\nto a value in the 5-10 second range."]}),"\n",(0,r.jsx)(n.admonition,{type:"important",children:(0,r.jsxs)(n.p,{children:["CoreDNS ",(0,r.jsx)(n.a,{href:"https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#stable-network-id",children:"caching timeout may need to be decreased"}),"\nfrom default 30 seconds to a value in the 5-10 second range"]})}),"\n",(0,r.jsxs)(n.p,{children:["If a stateless set is used recreated nodes will not have their persisted data and will start as blank nodes.\nThis can lead to data loss and higher network traffic volume due to more frequent\ndata synchronisation of both ",(0,r.jsx)(n.a,{href:"./quorum-queues",children:"quorum queues"}),"\nand ",(0,r.jsx)(n.a,{href:"./streams",children:"streams"})," on newly joining nodes."]}),"\n",(0,r.jsx)(n.h4,{id:"use-persistent-volumes",children:"Use Persistent Volumes"}),"\n",(0,r.jsxs)(n.p,{children:["How ",(0,r.jsx)(n.a,{href:"https://kubernetes.io/docs/concepts/storage/persistent-volumes/",children:"storage is configured"}),"\nis generally orthogonal to peer discovery. However, it does not make sense to run a stateful\ndata service such as RabbitMQ with ",(0,r.jsx)(n.a,{href:"./relocate",children:"node data directory"})," stored on a transient volume.\nUse of transient volumes can lead nodes to not have their persisted data after a restart.\nThis has the same consequences as with stateless sets covered above."]}),"\n",(0,r.jsxs)(n.h4,{id:"make-sure-etcrabbitmq-is-mounted-as-writeable",children:["Make Sure ",(0,r.jsx)(n.code,{children:"/etc/rabbitmq"})," is Mounted as Writeable"]}),"\n",(0,r.jsxs)(n.p,{children:["RabbitMQ nodes and images may need to update a file under ",(0,r.jsx)(n.code,{children:"/etc/rabbitmq"}),", the default ",(0,r.jsx)(n.a,{href:"./configure#config-location",children:"configuration file location"})," on Linux. This may involve configuration file generation\nperformed by the image used, ",(0,r.jsx)(n.a,{href:"./plugins#enabled-plugins-file",children:"enabled plugins file"})," updates,\nand so on."]}),"\n",(0,r.jsxs)(n.p,{children:["It is therefore highly recommended that ",(0,r.jsx)(n.code,{children:"/etc/rabbitmq"})," is mounted as writeable and owned by\nRabbitMQ's effective user (typically ",(0,r.jsx)(n.code,{children:"rabbitmq"}),")."]}),"\n",(0,r.jsx)(n.h4,{id:"use-parallel-podmanagementpolicy",children:"Use Parallel podManagementPolicy"}),"\n",(0,r.jsxs)(n.p,{children:[(0,r.jsx)(n.code,{children:'podManagementPolicy: "Parallel"'})," is the recommended option for RabbitMQ clusters."]}),"\n",(0,r.jsxs)(n.p,{children:["Because of ",(0,r.jsx)(n.a,{href:"./clustering#restarting",children:"how nodes rejoin their cluster"}),", ",(0,r.jsx)(n.code,{children:"podManagementPolicy"})," set to ",(0,r.jsx)(n.code,{children:"OrderedReady"}),"\ncan lead to a deployment deadlock with certain readiness probes:"]}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsx)(n.li,{children:"Kubernetes will expect the first node to pass a readiness probe"}),"\n",(0,r.jsx)(n.li,{children:"The readiness probe may require a fully booted node"}),"\n",(0,r.jsx)(n.li,{children:"The node will fully boot after it detects that its peers have come online"}),"\n",(0,r.jsx)(n.li,{children:"Kubernetes will not start any more pods until the first one boots"}),"\n",(0,r.jsx)(n.li,{children:"The deployment therefore is deadlocked"}),"\n"]}),"\n",(0,r.jsxs)(n.p,{children:[(0,r.jsx)(n.code,{children:'podManagementPolicy: "Parallel"'})," avoids this problem, and the Kubernetes peer discovery plugin\nthen deals with the ",(0,r.jsx)(n.a,{href:"./cluster-formation#initial-formation-race-condition",children:"natural race condition present during parallel cluster formation"}),"."]}),"\n",(0,r.jsx)(n.h4,{id:"use-most-basic-health-checks-for-rabbitmq-pod-readiness-probes",children:"Use Most Basic Health Checks for RabbitMQ Pod Readiness Probes"}),"\n",(0,r.jsxs)(n.p,{children:["A readiness probe that expects the node to be fully booted and have rejoined its cluster peers\ncan deadlock a deployment that restarts all RabbitMQ pods and relies on the ",(0,r.jsx)(n.code,{children:"OrderedReady"})," pod management policy.\nDeployments that use the ",(0,r.jsx)(n.code,{children:"Parallel"})," pod management policy\nwill not be affected."]}),"\n",(0,r.jsx)(n.p,{children:"One health check that does not expect a node to be fully booted and have schema tables synced is"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-bash",children:"# a very basic check that will succeed for the nodes that are currently waiting for\n# a peer to sync schema from\nrabbitmq-diagnostics ping\n"})}),"\n",(0,r.jsxs)(n.p,{children:["This basic check would allow the deployment to proceed and the nodes to eventually rejoin each other,\nassuming they are ",(0,r.jsx)(n.a,{href:"./upgrade",children:"compatible"}),"."]}),"\n",(0,r.jsxs)(n.p,{children:["See ",(0,r.jsx)(n.a,{href:"./clustering#restarting-schema-sync",children:"Schema Syncing from Online Peers"})," in the ",(0,r.jsx)(n.a,{href:"./clustering",children:"Clustering guide"}),"."]}),"\n",(0,r.jsx)(n.h3,{id:"examples",children:"Examples"}),"\n",(0,r.jsxs)(n.p,{children:["A minimalistic ",(0,r.jsx)(n.a,{href:"https://github.com/rabbitmq/diy-kubernetes-examples",children:"runnable example of Kubernetes peer discovery"}),"\nmechanism can be found on GitHub."]}),"\n",(0,r.jsx)(n.p,{children:"The example can be run using either MiniKube or Kind."}),"\n",(0,r.jsx)(n.h3,{id:"configuration-2",children:"Configuration"}),"\n",(0,r.jsxs)(n.p,{children:["To use Kubernetes for peer discovery, set the ",(0,r.jsx)(n.code,{children:"cluster_formation.peer_discovery_backend"}),"\nto ",(0,r.jsx)(n.code,{children:"k8s"})," or ",(0,r.jsx)(n.code,{children:"kubernetes"})," or its module name, ",(0,r.jsx)(n.code,{children:"rabbit_peer_discovery_k8s"}),"\n(note: the name of the module is slightly different from plugin name):"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = k8s\n\n# the backend can also be specified using its module name\n# cluster_formation.peer_discovery_backend = rabbit_peer_discovery_k8s\n\n# Kubernetes API hostname (or IP address). Default value is kubernetes.default.svc.cluster.local\ncluster_formation.k8s.host = kubernetes.default.example.local\n"})}),"\n",(0,r.jsx)(n.h4,{id:"kubernetes-api-endpoint",children:"Kubernetes API Endpoint"}),"\n",(0,r.jsx)(n.p,{children:"It is possible to configure Kubernetes API port and URI scheme:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = k8s\n\ncluster_formation.k8s.host = kubernetes.default.example.local\n# 443 is used by default\ncluster_formation.k8s.port = 443\n# https is used by default\ncluster_formation.k8s.scheme = https\n"})}),"\n",(0,r.jsx)(n.h4,{id:"kubernetes-api-access-token",children:"Kubernetes API Access Token"}),"\n",(0,r.jsxs)(n.p,{children:["Kubernetes token file path is configurable via ",(0,r.jsx)(n.code,{children:"cluster_formation.k8s.token_path"}),":"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = k8s\n\ncluster_formation.k8s.host = kubernetes.default.example.local\n# default value is /var/run/secrets/kubernetes.io/serviceaccount/token\ncluster_formation.k8s.token_path = /var/run/secrets/kubernetes.io/serviceaccount/token\n"})}),"\n",(0,r.jsx)(n.p,{children:"It must point to a local file that exists and is readable by RabbitMQ."}),"\n",(0,r.jsx)(n.h4,{id:"kubernetes-namespace",children:"Kubernetes Namespace"}),"\n",(0,r.jsxs)(n.p,{children:[(0,r.jsx)(n.code,{children:"cluster_formation.k8s.namespace_path"})," controls when the K8S namespace is loaded from:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = k8s\n\ncluster_formation.k8s.host = kubernetes.default.example.local\n\n# ...\n\n# Default value: /var/run/secrets/kubernetes.io/serviceaccount/namespace\ncluster_formation.k8s.namespace_path = /var/run/secrets/kubernetes.io/serviceaccount/namespace\n"})}),"\n",(0,r.jsxs)(n.p,{children:["Just like with the token path key, ",(0,r.jsx)(n.code,{children:"cluster_formation.k8s.namespace_path"})," must point to a local\nfile that exists and is readable by RabbitMQ."]}),"\n",(0,r.jsx)(n.h4,{id:"kubernetes-api-ca-certificate-bundle",children:"Kubernetes API CA Certificate Bundle"}),"\n",(0,r.jsxs)(n.p,{children:["Kubernetes API ",(0,r.jsx)(n.a,{href:"./ssl#certificates-and-keys",children:"CA certificate bundle"})," file path is\nconfigured using ",(0,r.jsx)(n.code,{children:"cluster_formation.k8s.cert_path"}),":"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = k8s\n\ncluster_formation.k8s.host = kubernetes.default.example.local\n\n# Where to load the K8S API access token from.\n# Default value: /var/run/secrets/kubernetes.io/serviceaccount/token\ncluster_formation.k8s.token_path = /var/run/secrets/kubernetes.io/serviceaccount/token\n\n# Where to load K8S API CA bundle file from. It will be used when issuing requests\n# to the K8S API using HTTPS.\n#\n# Default value: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt\ncluster_formation.k8s.cert_path = /var/run/secrets/kubernetes.io/serviceaccount/ca.crt\n"})}),"\n",(0,r.jsxs)(n.p,{children:["Just like with the token path key, ",(0,r.jsx)(n.code,{children:"cluster_formation.k8s.cert_path"})," must point to a local\nfile that exists and is readable by RabbitMQ."]}),"\n",(0,r.jsx)(n.h4,{id:"peer-node-pods-can-use-hostnames-or-ip-addresses",children:"Peer Node Pods Can Use Hostnames or IP Addresses"}),"\n",(0,r.jsxs)(n.p,{children:["When a list of peer nodes is computed from a list of pod containers returned by Kubernetes,\neither hostnames or IP addresses can be used. This is configurable using the\n",(0,r.jsx)(n.code,{children:"cluster_formation.k8s.address_type"})," key:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:'cluster_formation.peer_discovery_backend = k8s\n\ncluster_formation.k8s.host = kubernetes.default.example.local\n\ncluster_formation.k8s.token_path = /var/run/secrets/kubernetes.io/serviceaccount/token\ncluster_formation.k8s.cert_path = /var/run/secrets/kubernetes.io/serviceaccount/ca.crt\ncluster_formation.k8s.namespace_path = /var/run/secrets/kubernetes.io/serviceaccount/namespace\n\n# should result set use hostnames or IP addresses\n# of Kubernetes API-reported containers?\n# supported values are "hostname" and "ip"\ncluster_formation.k8s.address_type = hostname\n'})}),"\n",(0,r.jsxs)(n.p,{children:["Supported values are ",(0,r.jsx)(n.code,{children:"ip"})," or ",(0,r.jsx)(n.code,{children:"hostname"}),". ",(0,r.jsx)(n.code,{children:"hostname"})," is\nthe recommended option but has limitations: it can only be used with ",(0,r.jsx)(n.a,{href:"https://kubernetes.io/docs/tasks/run-application/run-replicated-stateful-application/#statefulset",children:"stateful sets"})," (also highly recommended)\nand ",(0,r.jsx)(n.a,{href:"https://kubernetes.io/docs/concepts/services-networking/service/#headless-services",children:"headless services"}),".\n",(0,r.jsx)(n.code,{children:"ip"})," is used by default for better compatibility."]}),"\n",(0,r.jsx)(n.h5,{id:"peer-node-pod-name-suffix",children:"Peer Node Pod Name Suffix"}),"\n",(0,r.jsxs)(n.p,{children:["It is possible to append a suffix to peer hostnames returned by Kubernetes using\n",(0,r.jsx)(n.code,{children:"cluster_formation.k8s.hostname_suffix"}),":"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = k8s\n\ncluster_formation.k8s.host = kubernetes.default.example.local\n\ncluster_formation.k8s.token_path = /var/run/secrets/kubernetes.io/serviceaccount/token\ncluster_formation.k8s.cert_path = /var/run/secrets/kubernetes.io/serviceaccount/ca.crt\ncluster_formation.k8s.namespace_path = /var/run/secrets/kubernetes.io/serviceaccount/namespace\n\n# no suffix is a
1ppended by default\ncluster_formation.k8s.hostname_suffix = rmq.eng.example.local\n"})}),"\n",(0,r.jsxs)(n.p,{children:["Service name is ",(0,r.jsx)(n.code,{children:"rabbitmq"})," by default but can be overridden using the\n",(0,r.jsx)(n.code,{children:"cluster_formation.k8s.service_name"})," key if needed:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:'cluster_formation.peer_discovery_backend = k8s\n\ncluster_formation.k8s.host = kubernetes.default.example.local\n\ncluster_formation.k8s.token_path = /var/run/secrets/kubernetes.io/serviceaccount/token\ncluster_formation.k8s.cert_path = /var/run/secrets/kubernetes.io/serviceaccount/ca.crt\ncluster_formation.k8s.namespace_path = /var/run/secrets/kubernetes.io/serviceaccount/namespace\n\n# overrides Kubernetes service name. Default value is "rabbitmq".\ncluster_formation.k8s.service_name = rmq-qa\n'})}),"\n",(0,r.jsx)(n.h2,{id:"peer-discovery-consul",children:"Peer Discovery Using Consul"}),"\n",(0,r.jsx)(n.h3,{id:"consul-peer-discovery-overview",children:"Consul Peer Discovery Overview"}),"\n",(0,r.jsxs)(n.p,{children:["A ",(0,r.jsx)(n.a,{href:"https://www.consul.io",children:"Consul"}),"-based discovery mechanism\nis available via ",(0,r.jsx)(n.a,{href:"https://github.com/rabbitmq/rabbitmq-server/tree/main/deps/rabbitmq_peer_discovery_consul",children:"a plugin"}),"."]}),"\n",(0,r.jsxs)(n.p,{children:["As with any ",(0,r.jsx)(n.a,{href:"./plugins",children:"plugin"}),", it must be enabled before it\ncan be used. For peer discovery plugins it means they must be ",(0,r.jsx)(n.a,{href:"./plugins#basics",children:"enabled"}),"\nor ",(0,r.jsx)(n.a,{href:"./plugins#enabled-plugins-file",children:"preconfigured"})," before first node boot:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-bash",children:"rabbitmq-plugins --offline enable rabbitmq_peer_discovery_consul\n"})}),"\n",(0,r.jsx)(n.p,{children:"The plugin supports Consul 0.8.0 and later versions."}),"\n",(0,r.jsxs)(n.p,{children:["Nodes register with Consul on boot and unregister when they\nleave. Prior to registration, nodes will attempt to acquire a\nlock in Consul to reduce the probability of a ",(0,r.jsx)(n.a,{href:"#initial-formation-race-condition",children:"race condition\nduring initial cluster formation"}),".\nWhen a node registers with Consul, it will set up a periodic ",(0,r.jsx)(n.a,{href:"https://www.consul.io/docs/agent/checks.html",children:"health\ncheck"})," for itself (more on this below)."]}),"\n",(0,r.jsx)(n.h3,{id:"configuration-3",children:"Configuration"}),"\n",(0,r.jsxs)(n.p,{children:["To use Consul for peer discovery, set the ",(0,r.jsx)(n.code,{children:"cluster_formation.peer_discovery_backend"}),"\nto ",(0,r.jsx)(n.code,{children:"consul"})," or its module name, ",(0,r.jsx)(n.code,{children:"rabbit_peer_discovery_consul"})," (note: the name of the module is\nslightly different from plugin name):"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\n# the backend can also be specified using its module name\n# cluster_formation.peer_discovery_backend = rabbit_peer_discovery_consul\n\n# Consul host (hostname or IP address). Default value is localhost\ncluster_formation.consul.host = consul.eng.example.local\n"})}),"\n",(0,r.jsx)(n.h4,{id:"consul-endpoint",children:"Consul Endpoint"}),"\n",(0,r.jsx)(n.p,{children:"It is possible to configure Consul port and URI scheme:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n# 8500 is used by default\ncluster_formation.consul.port = 8500\n# http is used by default\ncluster_formation.consul.scheme = http\n"})}),"\n",(0,r.jsx)(n.h4,{id:"consul-acl-token",children:"Consul ACL Token"}),"\n",(0,r.jsxs)(n.p,{children:["To configure ",(0,r.jsx)(n.a,{href:"https://www.consul.io/docs/guides/acl.html",children:"Consul ACL"})," token,\nuse ",(0,r.jsx)(n.code,{children:"cluster_formation.consul.acl_token"}),":"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\ncluster_formation.consul.acl_token = acl-token-value\n"})}),"\n",(0,r.jsx)(n.p,{children:'Service name (as registered in Consul) defaults to "rabbitmq" but can be overridden:'}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n# rabbitmq is used by default\ncluster_formation.consul.svc = rabbitmq\n"})}),"\n",(0,r.jsx)(n.h4,{id:"service-address",children:"Service Address"}
1),"\n",(0,r.jsx)(n.p,{children:"Service hostname (address) as registered in Consul will be fetched by peers\nand therefore must resolve on all nodes.\nThe hostname can be computed by the plugin or specified by the user. When computed automatically,\na number of nodes and OS properties can be used:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:["Hostname (as returned by ",(0,r.jsx)(n.code,{children:"gethostname(2)"}),")"]}),"\n",(0,r.jsxs)(n.li,{children:["Node name (without the ",(0,r.jsx)(n.code,{children:"rabbit@"})," prefix)"]}),"\n",(0,r.jsx)(n.li,{children:"IP address of an NIC (network controller interface)"}),"\n"]}),"\n",(0,r.jsxs)(n.p,{children:["When ",(0,r.jsx)(n.code,{children:"cluster_formation.consul.svc_addr_auto"})," is set to ",(0,r.jsx)(n.code,{children:"false"}),",\nservice name will be taken as is from ",(0,r.jsx)(n.code,{children:"cluster_formation.consul.svc_addr"}),".\nWhen it is set to ",(0,r.jsx)(n.code,{children:"true"}),", other options explained below come into play."]}),"\n",(0,r.jsxs)(n.admonition,{type:"important",children:[(0,r.jsxs)(n.p,{children:["Setting ",(0,r.jsx)(n.code,{children:"cluster_formation.consul.use_longname"})," to ",(0,r.jsx)(n.code,{children:"true"})," configures the Consul\npeer discovery plugin to use long node names. It does not, however, make RabbitMQ nodes use them."]}),(0,r.jsxs)(n.p,{children:["To make the nodes use long names, make sure the\n",(0,r.jsx)(n.a,{href:"./configure#supported-environment-variables",children:(0,r.jsx)(n.code,{children:"RABBITMQ_USE_LONGNAME"})})," is set to\n",(0,r.jsx)(n.code,{children:"true"})," for every cluster node, including any that might join the cluster later."]}),(0,r.jsxs)(n.p,{children:["See ",(0,r.jsx)(n.a,{href:"./clustering#node-names",children:"Node Names"})," in the Clustering guide to learn more."]})]}),"\n",(0,r.jsxs)(n.p,{children:["In the following example, the service address reported to Consul is\nhardcoded to ",(0,r.jsx)(n.code,{children:"hostname1.rmq.eng.example.local"})," instead of being computed automatically\nfrom the environment:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n\ncluster_formation.consul.svc = rabbitmq\n# do not compute service address, it will be specified below\ncluster_formation.consul.svc_addr_auto = false\n# service address, will be communicated to other nodes\ncluster_formation.consul.svc_addr = hostname1.rmq.eng.example.local\n# use long RabbitMQ node names?\ncluster_formation.consul.use_longname = true\n"})}),"\n",(0,r.jsxs)(n.p,{children:["In this example, the service address reported to Consul is\nparsed from node name (the ",(0,r.jsx)(n.code,{children:"rabbit@"})," prefix will be dropped):"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n\ncluster_formation.consul.svc = rabbitmq\n# do compute service address\ncluster_formation.consul.svc_addr_auto = true\n# compute service address using node name\ncluster_formation.consul.svc_addr_use_nodename = true\n# use long RabbitMQ node names?\ncluster_formation.consul.use_longname = true\n"})}),"\n",(0,r.jsxs)(n.p,{children:[(0,r.jsx)(n.code,{children:"cluster_formation.consul.svc_addr_use_nodename"})," is a boolean\nfield that instructs Consul peer discovery backend to compute service address\nusing RabbitMQ node name."]}),"\n",(0,r.jsx)(n.p,{children:"In the next example, the service address is\ncomputed using hostname as reported by the OS instead of node name:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n\ncluster_formation.consul.svc = rabbitmq\n# do compute service address\ncluster_formation.consul.svc_addr_auto = true\n# compute service address using host name and not node name\ncluster_formation.consul.svc_addr_use_nodename = false\n# use long RabbitMQ node names?\ncluster_formation.consul.use_longname = true\n"})}),"\n",(0,r.jsxs)(n.p,{children:["In the example below, the service address is\ncomputed by taking the IP address of a provided NIC, ",(0,r.jsx)(n.code,{children:"en0"}),":"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n\ncluster_formation.consul.svc = rabbitmq\n# do compute service address\ncluster_formation.consul.svc_addr_auto = true\n# compute service address using the IP address of a NIC, en0\ncluster_formation.consul.svc_addr_nic = en0\ncluster_formation.consul.svc_addr_use_nodename = false\n# use long RabbitMQ node names?\ncluster_formation.consul.use_longname = true\n"})}),"\n",(0,r.jsx)(n.h4,{id:"service-port",children:"Service Port"}),"\n",(0,r.jsxs)(n.p,{children:["Service port as registered in Consul can be overridden. This is only\nnecessary if RabbitMQ uses a ",(0,r.jsx)(n.a,{href:"./networking",children:"non-standard port"}
1),"\nfor client (technically AMQP 0-9-1 and AMQP 1.0) connections since default value is 5672."]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n# 5672 is used by default\ncluster_formation.consul.svc_port = 6674\n"})}),"\n",(0,r.jsx)(n.h4,{id:"service-tags-and-metadata",children:"Service Tags and Metadata"}),"\n",(0,r.jsxs)(n.p,{children:["It is possible to provide ",(0,r.jsx)(n.a,{href:"https://www.consul.io/docs/agent/./services",children:"Consul service tags"}),":"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:'cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n# Define tags for the RabbitMQ service: "qa" and "3.8"\ncluster_formation.consul.svc_tags.1 = qa\ncluster_formation.consul.svc_tags.2 = 3.8\n'})}),"\n",(0,r.jsxs)(n.p,{children:["It is possible to configure ",(0,r.jsx)(n.a,{href:"https://www.consul.io/docs/agent/./services",children:"Consul service metadata"}),",\nwhich is a map of string keys to string values with certain restrictions\n(see Consul documentation to learn more):"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n\n# Define metadata for the RabbitMQ service. Both keys and values have a\n# maximum length limit enforced by Consul. This can be used to provide additional\n# context about the service (RabbitMQ cluster) for operators or other tools.\ncluster_formation.consul.svc_meta.owner = team-xyz\ncluster_formation.consul.svc_meta.service = service-one\ncluster_formation.consul.svc_meta.stats_url = https://service-one.eng.megacorp.local/stats/\n"})}),"\n",(0,r.jsx)(n.h4,{id:"service-health-checks",children:"Service Health Checks"}),"\n",(0,r.jsxs)(n.p,{children:["When a node registers with Consul, it will set up a periodic ",(0,r.jsx)(n.a,{href:"https://www.consul.io/docs/agent/checks.html",children:"health check"}),"\nfor itself. Online nodes will periodically send a health check update to Consul to indicate the service\nis available. This interval can be configured:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n# health check interval (node TTL) in seconds\n# default: 30\ncluster_formation.consul.svc_ttl = 40\n"})}),"\n",(0,r.jsxs)(n.p,{children:["A node that failed its ",(0,r.jsx)(n.a,{href:"https://www.consul.io/docs/agent/checks.html",children:"health check"})," is considered\nto be in the warning state by Consul.\nSuch nodes can be automatically unregistered by Consul after a\nperiod of time (note: this is a separate interval value from\nthe TTL above). The period cannot be less than 60 seconds."]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n# health check interval (node TTL) in seconds\ncluster_formation.consul.svc_ttl = 30\n# how soon should nodes that fail their health checks be unregistered by Consul?\n# this value is in seconds and must not be lower than 60 (a Consul requirement)\ncluster_formation.consul.deregister_after = 90\n"})}),"\n",(0,r.jsxs)(n.p,{children:["Please see a section on ",(0,r.jsx)(n.a,{href:"#node-health-checks-and-cleanup",children:"automatic cleanup of nodes"})," below."]}),"\n",(0,r.jsxs)(n.p,{children:["Nodes in the warning state are excluded from peer discovery results\nby default. It is possible to opt into including them by setting\n",(0,r.jsx)(n.code,{children:"cluster_formation.consul.include_nodes_with_warnings"})," to\n",(0,r.jsx)(n.code,{children:"true"}),":"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n# health check interval (node TTL) in seconds\ncluster_formation.consul.svc_ttl = 30\n# include node in the warning state into discovery result set\ncluster_formation.consul.include_nodes_with_warnings = true\n"})}),"\n",(0,r.jsx)(n.h4,{id:"opting-out-of-regisration",children:"Opting Out of Regisration"}),"\n",(0,r.jsx)(n.p,{children:"If the configured backend supports registration,\nnodes unregister when they are instructed to stop."}),"\n",(0,r.jsxs)(n.p,{children:["It is possible to opt-out of registration completely with the config option\n",(0,r.jsx)(n.code,{children:"cluster_formation.registration.enabled"}),":"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.registration.enabled = false\n"})}),"\n",(0,r.jsxs)(n.p,{children:["When configured this way, the node has to be registered manually or using another mechanism,\ne.g. by a container orchestrator such as ",(0,r.jsx)(n.a,{href:"https://developer.hashicorp.com/nomad/integrations/hashicorp/rabbitmq",children:"Nomad"}),"."]}),"\n",(0,r.jsx)(n.h4,{id:"node-name-suffixes",children:"Node Name Suffixes"}),"\n",(0,r.jsxs)(n.p,{children:["If node name is computed and long node names are used, it is possible to\nappend a suffix to node names retrieved from Consul. The format is\n",(0,r.jsx)(n.code,{children:".node.{domain_suffix}"}),". This can be useful in environments with\nDNS conventions, e.g. when all service nodes\nare organised in a separate subdomain. Here's an example:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n\ncluster_formation.consul.svc = rabbitmq\n# do compute service address\ncluster_formation.consul.svc_addr_auto = true\n# compute service address using node name\ncluster_formation.consul.svc_addr_use_nodename = true\n# use long RabbitMQ node names?\ncluster_formation.consul.use_longname = true\n# append a suffix (node.rabbitmq.example.local) to node names retrieved from Consul\ncluster_formation.consul.domain_suffix = example.local\n"})}),"\n",(0,r.jsxs)(n.p,{children:["With this setup node names will be computed to ",(0,r.jsx)(n.code,{children:"[email protected]"}),"\ninstead of ",(0,r.jsx)(n.code,{children:"[email protected]"}),"."]}),"\n",(0,r.jsx)(n.h4,{id:"distributed-lock-acquisition",children:"Distributed Lock Acquisition"}),"\n",(0,r.jsx)(n.p,{children:"When a node tries to acquire a lock on boot and the lock is already taken,\nit will wait for the lock to become available for a limited amount of time. Default value is 300\nseconds but it can be configured:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\n# lock acquisition timeout in seconds\n# default: 300\n# cluster_formation.consul.lock_wait_time is an alias\ncluster_formation.consul.lock_t
1imeout = 60\n"})}),"\n",(0,r.jsxs)(n.p,{children:["Lock key prefix is ",(0,r.jsx)(n.code,{children:"rabbitmq"})," by default. It can also be overridden:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:'cluster_formation.peer_discovery_backend = consul\n\ncluster_formation.consul.host = consul.eng.example.local\ncluster_formation.consul.lock_timeout = 60\n# should the Consul key used for locking be prefixed with something\n# other than "rabbitmq"?\ncluster_formation.consul.lock_prefix = environments-qa\n'})}),"\n",(0,r.jsx)(n.h2,{id:"peer-discovery-etcd",children:"Peer Discovery Using Etcd"}),"\n",(0,r.jsx)(n.h3,{id:"etcd-peer-discovery-overview",children:"Etcd Peer Discovery Overview"}),"\n",(0,r.jsxs)(n.p,{children:["An ",(0,r.jsx)(n.a,{href:"https://etcd.io/",children:"etcd"}),"-based discovery mechanism\nis available via ",(0,r.jsx)(n.a,{href:"https://github.com/rabbitmq/rabbitmq-server/tree/main/deps/rabbitmq_peer_discovery_etcd",children:"a plugin"}),"."]}),"\n",(0,r.jsxs)(n.p,{children:["As of RabbitMQ ",(0,r.jsx)(n.code,{children:"3.8.4"}),", the plugin uses a v3 API, gRPC-based etcd client and\n",(0,r.jsx)(n.strong,{children:"requires etcd 3.4 or a later version"}),"."]}),"\n",(0,r.jsxs)(n.p,{children:["As with any ",(0,r.jsx)(n.a,{href:"./plugins",children:"plugin"}),", it must be enabled before it\ncan be used. For peer discovery plugins it means they must be ",(0,r.jsx)(n.a,{href:"./plugins#basics",children:"enabled"}),"\nor ",(0,r.jsx)(n.a,{href:"./plugins#enabled-plugins-file",children:"preconfigured"})," before first node boot:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-bash",children:"rabbitmq-plugins --offline enable rabbitmq_peer_discovery_etcd\n"})}),"\n",(0,r.jsx)(n.p,{children:"Nodes register with etcd on boot by creating a key in a conventionally named directory. The keys have\na short (say, a minute) expiration period. The keys are deleted when nodes stop cleanly."}),"\n",(0,r.jsxs)(n.p,{children:["Prior to registration, nodes will attempt to acquire a\nlock in etcd to reduce the probability of a ",(0,r.jsx)(n.a,{href:"#initial-formation-race-condition",children:"race condition\nduring initial cluster formation"}),"."]}),"\n",(0,r.jsxs)(n.p,{children:["Every node's key has an associated ",(0,r.jsx)(n.a,{href:"https://etcd.io/docs/v3.4.0/dev-guide/interacting_v3/#grant-leases",children:"lease"}),"\nwith a configurable TTL. Nodes keep their key's leases alive.\nIf a node loses connectivity and cannot update its lease, its key will be cleaned up by etcd after TTL expires.\nSuch nodes won't be discovered by newly joining nodes.\nIf configured, such nodes can be forcefully removed from the cluster."]}),"\n",(0,r.jsx)(n.h3,{id:"configuration-4",children:"Configuration"}),"\n",(0,r.jsx)(n.h4,{id:"etcd-endpoints-and-authentication",children:"etcd Endpoints and Authentication"}),"\n",(0,r.jsxs)(n.p,{children:["To use etcd for peer discovery, set the ",(0,r.jsx)(n.code,{children:"cluster_formation.peer_discovery_backend"}),"\nto ",(0,r.jsx)(n.code,{children:"etcd"})," or its module name, ",(0,r.jsx)(n.code,{children:"rabbit_peer_discovery_etcd"})," (note: the name of the module\nis slightly different from plugin name)."]}),"\n",(0,r.jsx)(n.p,{children:"The plugin requires a configured etcd endpoint for the plugin\nto connect to:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = etcd\n\n# the backend can also be specified using its module name\n# cluster_formation.peer_discovery_backend = rabbit_peer_discovery_etcd\n\n# etcd endpoints. This property is required or peer discovery won't be performed.\ncluster_formation.etcd.endpoints.1 = one.etcd.eng.example.local:2379\n"})}),"\n",(0,r.jsx)(n.p,{children:"It is possible to configure multiple etcd endpoints. The first randomly\nchosen one that the plugin can successfully connect to will be used."}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = etcd\n\ncluster_formation.etcd.endpoints.1 = one.etcd.eng.example.local:2379\ncluster_formation.etcd.endpoints.2 = two.etcd.eng.example.local:2479\ncluster_formation.etcd.endpoints.3 = three.etcd.eng.example.local:2579\n"})}),"\n",(0,r.jsxs)(n.p,{children:["If ",(0,r.jsx)(n.a,{href:"https://etcd.io/docs/v3.4.0/op-guide/authentication/",children:"authentication is enabled for etcd"}),", the plugin can be configured to use\na pair of credentials:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = etcd\n\ncluster_formation.etcd.endpoints.1 = one.etcd.eng.example.local:2379\ncluster_formation.etcd.endpoints.2 = two.etcd.eng.example.local:2479\ncluster_formation.etcd.endpoints.3 = three.etcd.eng.example.local:2579\n\ncluster_formation.etcd.username = rabbitmq\ncluster_formation.etcd.password = s3kR37\n"})}),"\n",(0,r.jsxs)(n.p,{children:["It is possible to use ",(0,r.jsx)(n.a,{href:"./configure#advanced-config-file",children:"advanced.config"})," file to ",(0,r.jsx)(n.a,{href:"./configure#configuration-encryption",children:"encrypt the password value"}),"\nlisted in the config. In this case all plugin settings must be moved to the advanced config:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-erlang",children:'%% advanced.config file\n[\n {rabbit,\n [{cluster_formation,\n [{peer_discovery_etcd, [\n {endpoints, [\n "one.etcd.eng.example.local:2379",\n "two.etcd.eng.example.local:2479",\n "three.etcd.eng.example.local:2579"\n ]},\n\n {etcd_prefix, "rabbitmq"},\n {cluster_name, "default"},\n\n {etcd_username, "etcd user"},\n {etcd_password, {encrypted, <<"cPAymwqmMnbPXXRVqVzpxJdrS8mHEKuo2V+3vt1u/fymexD9oztQ2G/oJ4PAaSb2c5N/hRJ2aqP/X0VAfx8xOQ==">>}\n }]\n }]\n }]\n },\n\n {config_entry_decoder, [\n {passphrase, <<"decryption key passphrase">>}\n ]}\n].\n'})}),"\n",(0,r.jsx)(n.h4,{id:"key-naming",children:"Key Naming"}
1),"\n",(0,r.jsxs)(n.p,{children:["Directories and keys used by the peer discovery mechanism follow a naming scheme.\nSince in etcd ",(0,r.jsx)(n.a,{href:"https://etcd.io/docs/v3.4.0/rfc/v3api/",children:"v3 API the key space is flat"}),",\na hardcoded prefix is used. It allows the plugin to predictably perform key range queries\nusing a well-known prefix:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"# for node presence keys\n/rabbitmq/discovery/{prefix}/clusters/{cluster name}/nodes/{node name}\n"})}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"# for registration lock keys\n/rabbitmq/locks/{prefix}/clusters/{cluster name}/registration\n"})}),"\n",(0,r.jsxs)(n.p,{children:["Here's an example of a key that would be used by node ",(0,r.jsx)(n.code,{children:"rabbit@hostname1"}),"\nwith default user-provided key prefix and cluster name:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"/rabbitmq/discovery/rabbitmq/clusters/default/nodes/rabbit@hostname1\n"})}),"\n",(0,r.jsx)(n.p,{children:'Default key prefix is simply "rabbitmq". It rarely needs overriding but that\'s\nsupported:'}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = etcd\n\ncluster_formation.etcd.endpoints.1 = one.etcd.eng.example.local:2379\ncluster_formation.etcd.endpoints.2 = two.etcd.eng.example.local:2479\ncluster_formation.etcd.endpoints.3 = three.etcd.eng.example.local:2579\n\n# rabbitmq is used by default\ncluster_formation.etcd.key_prefix = rabbitmq_discovery\n"})}),"\n",(0,r.jsx)(n.p,{children:"If multiple RabbitMQ clusters share an etcd installation, each cluster must use\na unique name:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:'cluster_formation.peer_discovery_backend = etcd\n\ncluster_formation.etcd.endpoints.1 = one.etcd.eng.example.local:2379\ncluster_formation.etcd.endpoints.2 = two.etcd.eng.example.local:2479\ncluster_formation.etcd.endpoints.3 = three.etcd.eng.example.local:2579\n\n# default name: "default"\ncluster_formation.etcd.cluster_name = staging\n'})}),"\n",(0,r.jsx)(n.h4,{id:"key-leases-and-ttl",children:"Key Leases and TTL"}),"\n",(0,r.jsx)(n.p,{children:"Key used for node registration will have a lease with a TTL associated with them.\nOnline nodes will periodically keep the leases alive (refresh). The TTL value can be configured:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = etcd\n\ncluster_formation.etcd.endpoints.1 = one.etcd.eng.example.local:2379\ncluster_formation.etcd.endpoints.2 = two.etcd.eng.example.local:2479\ncluster_formation.etcd.endpoints.3 = three.etcd.eng.example.local:2579\n\n# node TTL in seconds\n# default: 30\ncluster_formation.etcd.node_ttl = 40\n"})}),"\n",(0,r.jsx)(n.p,{children:"Key leases are updated periodically while the node is running and the plugin\nremains enabled."}),"\n",(0,r.jsx)(n.p,{children:"It is possible to forcefully remove the nodes that fail to refresh their keys from the cluster.\nThis is covered later in this guide."}),"\n",(0,r.jsx)(n.h4,{id:"locks",children:"Locks"}),"\n",(0,r.jsx)(n.p,{children:"When a node tries to acquire a lock on boot and the lock is already taken,\nit will wait for the lock to become available for a limited amount of time. Default value is 300\nseconds but it can be configured:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = etcd\n\ncluster_formation.etcd.endpoints.1 = one.etcd.eng.example.local:2379\ncluster_formation.etcd.endpoints.2 = two.etcd.eng.example.local:2479\n\n# lock acquisition timeout in seconds\n# default: 300\n# cluster_formation.consul.lock_wait_time is an alias\ncluster_formation.etcd.lock_timeout = 60\n"})}),"\n",(0,r.jsx)(n.h4,{id:"inspecting-keys",children:"Inspecting Keys"}),"\n",(0,r.jsxs)(n.p,{children:["In order to list all keys used by the etcd-based peer discovery mechanism, use ",(0,r.jsx)(n.code,{children:"etcdctl get"})," like so:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-bash",children:'etcdctl get --prefix=true "/rabbitmq"\n'})}),"\n",(0,r.jsx)(n.h4,{id:"tls",children:"TLS"}),"\n",(0,r.jsxs)(n.p,{children:["It is possible to configure the plugin to ",(0,r.jsx)(n.a,{href:"./ssl",children:"use TLS"}),' when connecting to etcd.\nTLS will be enabled if any of the TLS options listed below are configured, otherwise\nconnections will use "plain TCP" without TLS.']}),"\n",(0,r.jsxs)(n.p,{children:["The plugin acts as a TLS client. A ",(0,r.jsx)(n.a,{href:"./ssl#peer-verification",children:"trusted CA certificate"})," file must\nbe provided as well as a client certificate and private key pair:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = etcd\n\ncluster_formation.etcd.endpoints.1 = one.etcd.eng.example.local:2379\ncluster_formation.etcd.endpoints.2 = two.etcd.eng.example.local:2479\n\n# trusted CA certificate file path\ncluster_formation.etcd.ssl_options.cacertfile = /path/to/ca_certificate.pem\n# client certificate (public key) file path\ncluster_formation.etcd.ssl_options.certfile = /path/to/client_certificate.pem\n# client private key file path\ncluster_formation.etcd.ssl_options.keyfile = /path/to/client_key.pem\n\n# use TLSv1.2 for connections\ncluster_formation.etcd.ssl_options.versions.1 = tlsv1.2\n\n# enables peer verification (the plugin will verify the certificate chain of the server)\ncluster_formation.etcd.ssl_options.verify = verify_peer\ncluster_formation.etcd.ssl_options.fail_if_no_peer_cert = true\n"})}),"\n",(0,r.jsxs)(n.p,{children:["More ",(0,r.jsx)(n.a,{href:"./ssl",children:"TLS options"})," are supported such as cipher suites and\nclient-side session renegotiation options:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"cluster_formation.peer_discovery_backend = etcd\n\ncluster_formation.etcd.endpoints.1 = one.etcd.eng.example.local:2379\ncluster_formation.etcd.endpoints.2 = two.etcd.eng.example.local:2479\n\n# trusted CA certificate file path\ncluster_formation.etcd.ssl_options.cacertfile = /path/to/ca_certificate.pem\n# client certificate (public key) file path\ncluster_formation.etcd.ssl_options.certfile = /path/to/client_certificate.pem\n# client private key file path\ncluster_formation.etcd.ssl_options.keyfile = /path/to/client_key.pem\n\n# use TLSv1.2 for connections\ncluster_formation.etcd.ssl_options.versions.1 = tlsv1.2\n\n# enables peer verification (the plugin will verify the certificate chain of the server)\ncluster_formation.etcd.ssl_options.verify = verify_peer\ncluster_formation.etcd.ssl_options.fail_if_no_peer_cert = true\n\n# use secure session renegotiation\ncluster_formation.etcd.ssl_options.secure_renegotiate = true\n\n# Explicitly list enabled cipher suites. This can break connectivity\n# and is not necessary most of the time.\ncluster_formation.etcd.ssl_options.ciphers.1 = ECDHE-ECDSA-AES256-GCM-SHA384\ncluster_formation.etcd.ssl_options.ciphers.2 = ECDHE-RSA-AES256-GCM-SHA384\ncluster_formation.etcd.ssl_options.ciphers.3 = ECDH-ECDSA-AES256-GCM-SHA384\ncluster_formation.etcd.ssl_options.ciphers.4 = ECDH-RSA-AES256-GCM-SHA384\ncluster_formation.etcd.ssl_options.ciphers.5 = DHE-RSA-AES256-GCM-SHA384\ncluster_formation.etcd.ssl_options.ciphers.6 = DHE-DSS-AES256-GCM-SHA384\ncluster_formation.etcd.ssl_options.ciphers.7 = ECDHE-ECDSA-AES128-GCM-SHA256\ncluster_formation.etcd.ssl_options.ciphers.8 = ECDHE-RSA-AES128-GCM-SHA256\ncluster_formation.etcd.ssl_options.ciphers.9 = ECDH-ECDSA-AES128-GCM-SHA256\ncluster_formation.etcd.ssl_options.ciphers.10 = ECDH-RSA-AES128-GCM-SHA256\ncluster_formation.etcd.ssl_options.ciphers.11 = DHE-RSA-AES128-GCM-SHA256\ncluster_formation.etcd.ssl_options.ciphers.12 = DHE-DSS-AES128-GCM-SHA256\n"})}),"\n",(0,r.jsx)(n.h2,{id:"initial-formation-race-condition",children:"Race Conditions During Initial Cluster Formation"}),"\n",(0,r.jsx)(n.p,{children:"For successful cluster formation, only one node should form the cluster initially, that is,\nstart as a standalone node and initializing its database. If this was not the case, multiple\nclusters would be formed instead of just one, violating operator expectations."}),"\n",(0,r.jsx)(n.p,{children:"Consider a deployment where the entire cluster is provisioned at once and all nodes start in parallel.\nIn this case, a natural race condition occurs between the starting nodes.\nTo prevent multiple nodes forming separate cluster
1s, peer discovery backends try to acquire a lock when either\nforming the cluster (seeding) or joining a peer. What locks are used varies from backend to backend:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:["Classic config file, K8s, and AWS backends use a built-in ",(0,r.jsx)(n.a,{href:"https://erlang.org/doc/man/global.html#set_lock-3",children:"locking library"})," provided by the runtime"]}),"\n",(0,r.jsx)(n.li,{children:"The Consul peer discovery backend sets a lock in Consul"}),"\n",(0,r.jsx)(n.li,{children:"The etcd peer discovery backend sets a lock in etcd"}),"\n"]}),"\n",(0,r.jsxs)(n.p,{children:["The ",(0,r.jsx)(n.a,{href:"#peer-discovery-dns",children:"DNS peer discovery mechanism"})," does not use locking and requires a random startup delay\nto be injected by the deployment tooling. A random value in the 1 to 20 second range is recommended\nfor most environments. In the environments where nodes are dynamically registered\nwith the DNS system and it can take time, use a broader range, such as 1 to 60 seconds."]}),"\n",(0,r.jsx)(n.h2,{id:"node-health-checks-and-cleanup",children:"Node Health Checks and Forced Removal"}),"\n",(0,r.jsxs)(n.p,{children:["Nodes in clusters formed using peer discovery can fail, become unavailable or be permanently\nremoved (decommissioned). Some operators may want such nodes to be automatically removed\nfrom the cluster after a period of time. Such automated forced removal also can produce\nunforeseen side effects, so RabbitMQ does not enforce this behavior. It ",(0,r.jsx)(n.strong,{children:"should be used\nwith great care"})," and only if the side effects are fully understood and considered."]}),"\n",(0,r.jsx)(n.p,{children:'For example, consider a cluster that uses the AWS backend configured to use autoscaling group membership.\nIf an EC2 instance in that group fails and is later re-created as a new node, its original "incarnation"\nwill be considered a separate, now permanently unavailable node in the same cluster.'}),"\n",(0,r.jsx)(n.p,{children:"With peer discovery backends that offer dynamic node management (as opposed to, say, a fixed list of nodes\nin the configuration file), such unknown nodes can be logged or forcefully removed from the cluster."}),"\n",(0,r.jsx)(n.p,{children:"They are"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsx)(n.li,{children:(0,r.jsx)(n.a,{href:"#peer-discovery-aws",children:"AWS (EC2)"})}),"\n",(0,r.jsx)(n.li,{children:(0,r.jsx)(n.a,{href:"#peer-discovery-k8s",children:"Kubernetes"})}),"\n",(0,r.jsx)(n.li,{children:(0,r.jsx)(n.a,{href:"#peer-discovery-consul",children:"Consul"})}),"\n",(0,r.jsx)(n.li,{children:(0,r.jsx)(n.a,{href:"#peer-discovery-etcd",children:"etcd"})}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:"Forced node removal can be dangerous and should be carefully considered. For example,\na node that's temporarily unavailable but will be rejoining (or recreated with its\npersistent storage re-attached from its previous incarnation) can be kicked\nout of the cluster permanently by automatic cleanup, thus failing to rejoin."}),"\n",(0,r.jsx)(n.p,{children:"Before enabling the configuration keys covered below make sure that a compatible\npeer discovery plugin is enabled. If that's not the case the node will report\nthe settings to be unknown and will fail to start."}),"\n",(0,r.jsxs)(n.p,{children:["To log warnings for the unknown nodes,\n",(0,r.jsx)(n.code,{children:"cluster_formation.node_cleanup.only_log_warning"})," should be set to\n",(0,r.jsx)(n.code,{children:"true"}),":"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"# Don't remove cluster members unknown to the peer discovery backend but log\n# warnings.\n#\n# This setting can only be used if a compatible peer discovery plugin is enabled.\ncluster_formation.node_cleanup.only_log_warning = true\n"})}),"\n",(0,r.jsx)(n.p,{children:"This is the default behavior."}),"\n",(0,r.jsxs)(n.p,{children:["To forcefully delete the unknown nodes from the cluster,\n",(0,r.jsx)(n.code,{children:"cluster_formation.node_cleanup.only_log_warning"})," should be set to\n",(0,r.jsx)(n.code,{children:"false"}),"."]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"# Forcefully remove cluster members unknown to the peer discovery backend. Once removed,\n# the nodes won't be able to rejoin. Use this mode with great c
1are!\n#\n# This setting can only be used if a compatible peer discovery plugin is enabled.\ncluster_formation.node_cleanup.only_log_warning = false\n"})}),"\n",(0,r.jsx)(n.p,{children:"Note that this option should be used with care, in particular\nwith discovery backends other than AWS."}),"\n",(0,r.jsx)(n.p,{children:"The cleanup checks are performed periodically. The interval is 60 seconds\nby default and can be overridden:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"# perform the check every 90 seconds\ncluster_formation.node_cleanup.interval = 90\n"})}),"\n",(0,r.jsxs)(n.p,{children:["Some backends (Consul, etcd) support node health checks or TTL. These checks\nshould not to be confused with ",(0,r.jsx)(n.a,{href:"./monitoring#health-checks",children:"monitoring health checks"}),".\nThey allow peer discovery services (such as etcd or Consul) keep track of what\nnodes are still around (have checked in recently)."]}),"\n",(0,r.jsx)(n.p,{children:"With service discovery health checks, nodes set a TTL on their keys and/or periodically\nnotify their respective discovery service that they are still present. If no notifications\nfrom a node come in after a period of time, the node's key will eventually expire (with\nConsul, such nodes will be considered to be in a warning state)."}),"\n",(0,r.jsx)(n.p,{children:"With etcd, such nodes will no longer show up in discovery results. With Consul,\nthey can either be removed (deregistered) or their warning state can be\nreported. Please see documentation for those backends to learn more."}),"\n",(0,r.jsx)(n.p,{children:"Automatic cleanup of absent nodes makes most sense in environments where failed/discontinued nodes\nwill be replaced with brand new ones (including cases when persistent storage won't be re-attached)."}),"\n",(0,r.jsxs)(n.p,{children:["When automatic node cleanup is deactivated (switched to the warning mode), operators have to\nexplicitly remove absent cluster nodes using ",(0,r.jsx)(n.a,{href:"./cli",children:(0,r.jsx)(n.code,{children:"rabbitmqctl forget_cluster_node"})}),"."]}),"\n",(0,r.jsx)(n.h3,{id:"negative-side-effects-of-automatic-removal",children:"Negative Side Effects of Automatic Removal"}),"\n",(0,r.jsxs)(n.p,{children:["Automatic node removal has a number of negative side effects operators should be aware of.\nA node that's temporarily unreachable, for example, because it's lost connectivity\nto the rest of the network or its VM was temporarily suspended, will be removed and will\nthen come back. Such node won't be able to ",(0,r.jsx)(n.a,{href:"#rejoining",children:"rejoin its cluster"})," and will\nlog a similar message:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{children:"Node '[email protected]' thinks it's clustered with node '[email protected]', but '[email protected]' disagrees\n"})}),"\n",(0,r.jsxs)(n.p,{children:["In addition, such nodes can begin to fail their ",(0,r.jsx)(n.a,{href:"./monitoring#health-checks",children:"monitoring health checks"}),',\nas they would be in a permanent "partitioned off" state. Even though such nodes might have been\nreplaced with a new one and the cluster would be operating as expected, such automatically removed\nand replaced nodes can produce monitoring false positives.']}),"\n",(0,r.jsx)(n.p,{children:"The list of side effects is not limited to those two scenarios but they all have the same\nroot cause: an automatically removed node can come back without realising that it's been kicked out\nof its cluster. Monitoring systems and operators won't be immediately aware of that event either."}),"\n",(0,r.jsx)(n.h2,{id:"discovery-retries",children:"Peer Discovery Failures and Retries"}),"\n",(0,r.jsxs)(n.p,{children:["In latest releases if a peer discovery attempt fail, it will be retried up to a certain number\nof times with a delay between each attempt. This is similar to the peer sync retries nodes\nperform ",(0,r.jsx)(n.a,{href:"./clustering#restarting",children:"when they come online after a restart"}),"."]}),"\n",(0,r.jsxs)(n.p,{children:["For example, with the ",(0,r.jsx)(n.a,{href:"#peer-discovery-k8s",children:"Kubernetes peer discovery mechanism"})," this means that\nKubernetes API requests that list pods will be retried should they fail. With the AWS mechanism,\nEC2 API requests are retried, and so on."]}),"\n",(0,r.jsxs)(n.p,{children:["Such retries by no means handle every possible failure scenario but they improve the resilience\nof peer discovery and thus cluster and node deployments in practice. However, if clustered\nnodes ",(0,r.jsx)(n.a,{href:"./clustering#erlang-cookie",children:"fail to authenticate"})," with each other, retries\nwill simply merely the inevitable failure of cluster formation."]}),"\n",(0,r.jsxs)(n.p,{children:["Nodes that fail to perform peer discovery will ",(0,r.jsx)(n.a,{href:"./logging",children:"log"})," their remaining recovery attempts:"]}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{children:"2020-06-27 06:35:36.426 [error] <0.277.0> Trying to join discovered peers failed. Will retry after a delay of 500 ms, 4 retries left...\n2020-06-27 06:35:36.928 [warning] <0.277.0> Could not auto-cluster with node rabbit@hostname2: {badrpc,nodedown}\n2020-06-27 06:35:36.930 [warning] <0.277.0> Could not auto-cluster with node rabbit@hostname3: {badrpc,nodedown}\n2020-06-27 06:35:36.930 [error] <0.277.0> Trying to join discovered peers failed. Will retry after a delay of 500 ms, 3 retries left...\n2020-06-27 06:35:37.432 [warning] <0.277.0> Could not auto-cluster with node rabbit@hostname2: {badrpc,nodedown}\n2020-06-27 06:35:37.434 [warning] <0.277.0> Could not auto-cluster with node rabbit@hostname3: {badrpc,nodedown}\n"})}),"\n",(0,r.jsxs)(n.p,{children:["If a node fails to perform peer discovery and exhausts all retries, ",(0,r.jsx)(n.a,{href:"./logging#debug-logging",children:"enable debug logging"})," is highly recommended for ",(0,r.jsx)(n.a,{href:"#troubleshooting",children:"troubleshooting"}),"."]}),"\n",(0,r.jsx)(n.p,{children:"The number of retries and the delay can be configured:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"# These are the default values\n\n# Retry peer discovery operations up to ten times\ncluster_formation.discovery_retry_limit = 10\n\n# 500 milliseconds\ncluster_formation.discovery_retry_interval = 500\n"})}),"\n",(0,r.jsx)(n.p,{children:"The defaults cover five seconds of unavailability of services, API endpoints or nodes\ninvolved in peer discovery. These values are sufficient to cover sporadic failures.\nThey will require increasing in environments where dependent services (DNS, etcd, Consul, etc)\nmay be provisioned concurrently with RabbitMQ cluster deployment and thus can become\navailable only after a period of time."}),"\n",(0,r.jsx)(n.h2,{id:"http-proxy-settings",children:"HTTP Proxy Settings"}),"\n",(0,r.jsx)(n.p,{children:"Peer discovery mechanisms that use HTTP to interact with its dependencies (e.g. AWS, Consul\nand etcd ones) can proxy their requests using an HTTP proxy."}),"\n",(0,r.jsx)(n.p,{children:"There are separate proxy settings for HTTP and HTTPS:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"# example HTTP and HTTPS proxy servers, values in your environment\n# will vary\ncluster_formation.proxy.http_proxy = 192.168.0.98\ncluster_formation.proxy.https_proxy = 192.168.0.98\n"})}),"\n",(0,r.jsx)(n.p,{children:"Some hosts can be excluded from proxying, e.g. the link-local AWS instance metadata\nIP address:"}),"\n",(0,r.jsx)(n.pre,{children:(0,r.jsx)(n.code,{className:"language-ini",children:"# example HTTP and HTTPS proxy servers, values in your environment\n# will vary\ncluster_formation.proxy.http_proxy = 192.168.0.98\ncluster_formation.proxy.https_proxy = 192.168.0.98\n\n# requests to these hosts won't go via proxy\ncluster_formation.proxy.proxy_exclusions.1 = 169.254.169.254\ncluster_formation.proxy.proxy_exclusions.2 = excluded.example.local\n"})}),"\n",(0,r.jsx)(n.h2,{id:"troubleshooting",children:"Troubleshooting"}),"\n",(0,r.jsxs)(n.p,{children:["The peer discovery subsystem and individual mechanism implementations log important\ndiscovery procedure steps at the ",(0,r.jsx)(n.code,{children:"info"})," log level. More extensive logging\nis available at the ",(0,r.jsx)(n.code,{children:"debug"})," level. Mechanisms that depend on external services\naccessible over HTTP will log all outgoing HTTP requests and response codes at ",(0,r.jsx)(n.code,{children:"debug"})," level.\nSee the ",(0,r.jsx)(n.a,{href:"./logging",children:"logging guide"})," for more information about logging configuration."]}),"\n",(0,r.jsx)(n.p,{children:"If the log does not contain any entries that demonstrate peer discovery progress, for example, the list\nof nodes retrieved by the mechanism or clustering attempts, it may mean that the node already has\nan initialised data directory or is already a member of the cluster. In those cases peer discovery\nwon't be performed."}),"\n",(0,r.jsxs)(n.p,{children:["Peer discovery relies on inter-node network connectivity and successful authentication via a shared\nsecret. Verifying that nodes can communicate with one another and use the expected Erlang cookie value (that's also identical across all cluster nodes).\nSee the main ",(0,r.jsx)(n.a,{href:"./clustering",children:"Clustering guide"})," for more information."]}),"\n",(0,r.jsxs)(n.p,{children:["A methodology for network connectivity troubleshooting as well as commonly used\ntools are covered in the ",(0,r.jsx)(n.a,{href:"./troubleshooting-networking",children:"Troubleshooting Network Connectivity"})," guide."]})]})}function h(e={}){let{wrapper:n}={...(0,t.R)(),...e.components};return n?(0,r.jsx)(n,{...e,children:(0,r.jsx)(d,{...e})}):d(e)}},28453(e,n,s){s.d(n,{R:()=>o,x:()=>a});var i=s(96540);let r={},t=i.createContext(r);function o(e){let n=i.useContext(t);return i.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function a(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(r):e.components||r:o(e.components),i.createElement(t.Provider,{value:n},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.