PageSourceSearch

https://llm-d.ai/assets/js/1e065143.108d1493.js

js llm-d.ai collected 2026-10-01 10:43:24 UTC 7,850 bytes, 1 lines download raw bytes

1"use strict";(globalThis.webpackChunkllm_d_website=globalThis.webpackChunkllm_d_website||[]).push([[42806],{29889(e,n,r){r.r(n),r.d(n,{assets:()=>l,contentTitle:()=>c,default:()=>h,frontMatter:()=>s,metadata:()=>t,toc:()=>d});const t=JSON.parse('{"id":"architecture/core/inferencepool","title":"InferencePool","description":"The InferencePool is the central resource that bridges the gap between the Gateway, the Endpoint Policy Provider (EPP), and the collection of model server instances. It serves as the source of truth for both endpoint discovery and service mesh/gateway integration.","source":"@site/docs/architecture/core/inferencepool.md","sourceDirName":"architecture/core","slug":"/architecture/core/inferencepool","permalink":"/docs/dev/architecture/core/inferencepool","draft":false,"unlisted":false,"editUrl":"https://github.com/llm-d/llm-d/edit/main/docs/architecture/core/inferencepool.md","tags":[],"version":"current","frontMatter":{},"sidebar":"docsSidebar","previous":{"title":"Architecture","permalink":"/docs/dev/architecture/"},"next":{"title":"llm-d Router","permalink":"/docs/dev/architecture/core/router/"}}');var o=r(74848),i=r(28453);const s={},c="InferencePool",l={},d=[{value:"Functional Overview",id:"functional-overview",level:2},{value:"Architecture and Relations",id:"architecture-and-relations",level:2},{value:"1. Endpoint Discovery (EPP Perspective)",id:"1-endpoint-discovery-epp-perspective",level:3},{value:"2. Gateway Integration (Controller Perspective)",id:"2-gateway-integration-controller-perspective",level:3},{value:"Key Relationships",id:"key-relationships",level:2}];function a(e){const n={code:"code",h1:"h1",h2:"h2",h3:"h3",header:"header",img:"img",li:"li",ol:"ol",p:"p",picture:"picture",source:"source",strong:"strong",ul:"ul",...(0,i.R)(),...e.components};return(0,o.jsxs)(o.Fragment,{children:[(0,o.jsx)(n.header,{children:(0,o.jsx)(n.h1,{id:"inferencepool",children:"InferencePool"})}),"\n",(0,o.jsxs)(n.p,{children:["The ",(0,o.jsx)(n.code,{children:"InferencePool"})," is the central resource that bridges the gap between the Gateway, the Endpoint Policy Provider (EPP), and the collection of model server instances. It serves as the source of truth for both endpoint discovery and service mesh/gateway integration."]}),"\n",(0,o.jsx)(n.h2,{id:"functional-overview",children:"Functional Overview"}),"\n",(0,o.jsxs)(n.p,{children:["The ",(0,o.jsx)(n.code,{children:"InferencePool"})," performs two primary roles in the inference infrastructure:"]}),"\n",(0,o.jsxs)(n.ol,{children:["\n",(0,o.jsxs)(n.li,{children:[(0,o.jsx)(n.strong,{children:"Endpoint Discovery for the EPP:"})," It defines how the EPP should find and monitor the model server Pods that are eligible to serve requests."]}),"\n",(0,o.jsxs)(n.li,{children:[(0,o.jsx)(n.strong,{children:"Service Integration for the Gateway:"})," It provides the necessary metadata for the Gateway controller to locate the EPP and connect it to the proxy as an external processing (",(0,o.jsx)(n.code,{children:"ext-proc"}),") service."]}),"\n"]}),"\n",(0,o.jsx)(n.h2,{id:"architecture-and-relations",children:"Architecture and Relations"}),"\n",(0,o.jsxs)(n.p,{children:["The following diagram visualizes how the ",(0,o.jsx)(n.code,{children:"InferencePool"})," resource is involved in the control path of both the EPP and Gateway Controller:"]}),"\n",(0,o.jsxs)(n.p,{align:"center",children:["\n  ",(0,o.jsxs)(n.picture,{children:["\n    ",(0,o.jsx)(n.source,{media:"(prefers-color-scheme: dark)"}),"\n    ",(0,o.jsx)(n.img,{src:"/img/docs/assets/gateway-design.svg",alt:"InferencePool"}),"\n  "]}),"\n"]}),"\n",(0,o.jsx)(n.h3,{id:"1-endpoint-discovery-epp-perspective",children:"1. Endpoint Discovery (EPP Perspective)"}),"\n",(0,o.jsxs)(n.p,{children:["The EPP uses the ",(0,o.jsx)(n.code,{children:"InferencePool"})," to discover which pods it can pick from."]}),"\n",(0,o.jsxs)(n.ul,{children:["\n",(0,o.jsxs)(n.li,{children:[(0,o.jsx)(n.strong,{children:"Selector-based Discovery:"})," The ",(0,o.jsx)(n.code,{children:"InferencePool"})," defines a ",(0,o.jsx)(n.code,{children:"selector"})," (label matching). The EPP watches for Pods that match these labels within the same namespace."]}),"\n",(0,o.jsxs)(n.li,{children:[(0,o.jsx)(n.strong,{children:"Dynamic Membership:"})," As model server Pods are scaled up or down, or as their readiness state changes, the EPP automatically updates its internal list of healthy candidates."]}),"\n",(0,o.jsxs)(n.li,{children:[(0,o.jsx)(n.strong,{children:"Port Mapping:"})," The ",(0,o.jsx)(n.code,{children:"targetPorts"})," in the ",(0,o.jsx)(n.code,{children:"InferencePool"})," tell the EPP which ports on the discovered Pods are listening for inference traffic (e.g., port 8000 for vLLM)."]}),"\n"]}),"\n",(0,o.jsx)(n.h3,{id:"2-gateway-integration-controller-perspective",children:"2. Gateway Integration (Controller Perspective)"}),"\n",(0,o.jsxs)(n.p,{children:["When an ",(0,o.jsx)(n.code,{children:"InferencePool"})," is used as a backendRef in an ",(0,o.jsx)(n.code,{children:"HTTPRoute"}),", the Gateway controller uses the resource to configure the underlying proxy."]}),"\n",(0,o.jsxs)(n.ul,{children:["\n",(0,o.jsxs)(n.li,{children:[(0,o.jsx)(n.strong,{children:"EPP Connectivity:"})," The ",(0,o.jsx)(n.code,{children:"endpointPickerRef"})," (or ",(0,o.jsx)(n.code,{children:"extensionRef"}),") in the ",(0,o.jsx)(n.code,{children:"InferencePool"})," points to the EPP service. The Gateway controller uses this information to configure the proxy's ",(0,o.jsx)(n.code,{children:"ext_proc"})," filter, ensuring that every request directed to the pool is first processed by the EPP."]}),"\n",(0,o.jsxs)(n.li,{children:[(0,o.jsx)(n.strong,{children:"Routing Logic:"}),' The proxy is configured to "park" the request and wait for the EPP\'s decision. The EPP then instructs the proxy\u2014via the ',(0,o.jsx)(n.code,{children:"ext_proc"})," protocol\u2014on which specific Pod IP from the discovered pool should receive the request."]}),"\n",(0,o.jsxs)(n.li,{children:[(0,o.jsx)(n.strong,{children:"Failure Handling:"})," The ",(0,o.jsx)(n.code,{children:"failureMode"})," defined in the ",(0,o.jsx)(n.code,{children:"InferencePool"})," (e.g., ",(0,o.jsx)(n.code,{children:"FailOpen"})," or ",(0,o.jsx)(n.code,{children:"FailClose"}),") tells the Gateway controller how to configure the proxy's behavior if the EPP becomes unresponsive."]}),"\n"]}),"\n",(0,o.jsx)(n.h2,{id:"key-relationships",children:"Key Relationships"}),"\n",(0,o.jsxs)(n.ul,{children:["\n",(0,o.jsxs)(n.li,{children:[(0,o.jsx)(n.strong,{children:"One-to-One Mapping:"})," Typically, one ",(0,o.jsx)(n.code,{children:"InferencePool"})," corresponds to one logical deployment of a model (e.g., Gemma4) and is served by one EPP deployment."]}),"\n",(0,o.jsxs)(n.li,{children:[(0,o.jsx)(n.strong,{children:"Decoupled Scaling:"})," The model servers can scale independently of the EPP. The ",(0,o.jsx)(n.code,{children:"InferencePool"})," ensures the EPP is always aware of the current set of available endpoints."]}),"\n",(0,o.jsxs)(n.li,{children:[(0,o.jsx)(n.strong,{children:"Namespace Scoped:"})," All discovery and references (Pods, EPP Service, and the InferencePool itself) are strictly contained within the same Kubernetes namespace to maintain security and isolation boundaries."]}),"\n"]})]})}function h(e={}){const{wrapper:n}={...(0,i.R)(),...e.components};return n?(0,o.jsx)(n,{...e,children:(0,o.jsx)(a,{...e})}):a(e)}},28453(e,n,r){r.d(n,{R:()=>s,x:()=>c});var t=r(96540);const o={},i=t.createContext(o);function s(e){const n=t.useContext(i);return t.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function c(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(o):e.components||o:s(e.components),t.createElement(i.Provider,{value:n},e.children)}}}]);

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.