1"use strict";(globalThis.webpackChunkllm_d_website=globalThis.webpackChunkllm_d_website||[]).push([[33016],{71871(e,r,n){n.r(r),n.d(r,{assets:()=>d,contentTitle:()=>c,default:()=>h,frontMatter:()=>l,metadata:()=>t,toc:()=>o});const t=JSON.parse('{"id":"architecture/core/model-servers","title":"Model Servers","description":"The model server is the component that runs inference on a model. llm-d supports vLLM, SGLang, and TensorRT-LLM (trtllm-serve) as model server backends.","source":"@site/docs/architecture/core/model-servers.md","sourceDirName":"architecture/core","slug":"/architecture/core/model-servers","permalink":"/docs/dev/architecture/core/model-servers","draft":false,"unlisted":false,"editUrl":"https://github.com/llm-d/llm-d/edit/main/docs/architecture/core/model-servers.md","tags":[],"version":"current","frontMatter":{},"sidebar":"docsSidebar","previous":{"title":"Configuration","permalink":"/docs/dev/architecture/core/router/epp/configuration"},"next":{"title":"Disaggregation","permalink":"/docs/dev/architecture/advanced/disaggregation/"}}');var s=n(74848),i=n(28453);const l={},c="Model Servers",d={},o=[{value:"Functionality",id:"functionality",level:2},{value:"EPP <-> Model Server Protocol",id:"epp---model-server-protocol",level:2},{value:"Metrics Reporting",id:"metrics-reporting",level:3},{value:"LoRA Adapter Serving",id:"lora-adapter-serving",level:3},{value:"Prefix Cache Reuse",id:"prefix-cache-reuse",level:3},{value:"Health Checks",id:"health-checks",level:3},{value:"Further Reading",id:"further-reading",level:2}];function a(e){const r={a:"a",admonition:"admonition",code:"code",h1:"h1",h2:"h2",h3:"h3",header:"header",img:"img",li:"li",p:"p",picture:"picture",pre:"pre",source:"source",strong:"strong",table:"table",tbody:"tbody",td:"td",th:"th",thead:"thead",tr:"tr",ul:"ul",...(0,i.R)(),...e.components};return(0,s.jsxs)(s.Fragment,{children:[(0,s.jsx)(r.header,{children:(0,s.jsx)(r.h1,{id:"model-servers",children:"Model Servers"})}),"\n",(0,s.jsxs)(r.p,{children:["The model server is the component that runs inference on a model. llm-d supports vLLM, SGLang, and TensorRT-LLM (",(0,s.jsx)(r.code,{children:"trtllm-serve"}),") as model server backends."]}),"\n",(0,s.jsx)(r.h2,{id:"functionality",children:"Functionality"}),"\n",(0,s.jsx)(r.p,{children:"A model server loads a model onto one or more accelerators (GPUs, TPUs, etc.) and exposes a supported API, such as OpenAI-compatible API, for inference requests. In the llm-d architecture, model servers are the compute layer -- they execute the actual prefill and decode steps that generate tokens."}),"\n",(0,s.jsx)(r.p,{children:"Model servers are the lowest layer in the llm-d stack:"}),"\n",(0,s.jsxs)(r.p,{align:"center",children:["\n ",(0,s.jsxs)(r.picture,{children:["\n ",(0,s.jsx)(r.source,{media:"(prefers-color-scheme: dark)"}),"\n ",(0,s.jsx)(r.img,{src:"/img/docs/assets/basic-architecture.svg",alt:"Architecture"}),"\n "]}),"\n"]}),"\n",(0,s.jsxs)(r.p,{children:["Model servers are deployed independently from the rest of the llm-d stack. They join an ",(0,s.jsx)(r.code,{children:"InferencePool"})," automatically via Kubernetes label selectors, and the EPP begins routing traffic to them once they are healthy."]}),"\n",(0,s.jsx)(r.p,{children:"Key responsibilities:"}),"\n",(0,s.jsxs)(r.ul,{children:["\n",(0,s.jsxs)(r.li,{children:[(0,s.jsx)(r.strong,{children:"Serve inference requests"})," via an OpenAI-compatible API (",(0,s.jsx)(r.code,{children:"/v1/completions"}),", ",(0,s.jsx)(r.code,{children:"/v1/chat/completions"}),")"]}),"\n",(0,s.jsxs)(r.li,{children:[(0,s.jsx)(r.strong,{children:"Expose metrics"})," (KV-cache utilization, queue depth, active requests) that the EPP uses for intelligent scheduling"]}),"\n",(0,s.jsxs)(r.li,{children:[(0,s.jsx)(r.strong,{children:"Manage KV-cache"})," on GPU memory, including prefix caching for repeated prompt prefixes"]}),"\n",(0,s.jsxs)(r.li,{children:[(0,s.jsx)(r.strong,{children:"Support parallelism strategies"})," such as Tensor Parallelism (TP), Data Parallelism (DP), and Expert Parallelism (EP) for large models"]}),"\n"]}),"\n",(0,s.jsx)(r.h2,{id:"epp---model-server-protocol",children:"EPP <-> Model Server Protocol"}),"\n",(0,s.jsx)(r.p,{children:"This is the protocol between the EPP and the model servers."}),"\n",(0,s.jsx)(r.h3,{id:"metrics-reporting",children:"Metrics Reporting"}),"\n",(0,s.jsx)(r.p,{children:"By default, the EPP is configured to scrape metrics from the model servers to make optimal request scheduling\ndecisions. In this mode of operation, the model servers MUST provide the following metrics via a Prometheus endpoint. The exact\nmetric names don't necessarily need to be the same as the recommended names here, however the\nmetric types and semantics MUST follow this doc."}),"\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n",(0,s.jsxs)(r.table,{children:[(0,s.jsx)(r.thead,{children:(0,s.jsxs)(r.tr,{children:[(0,s.jsx)(r.th,{children:"Metric"}),(0,s.jsx)(r.th,{children:"Type"}),(0,s.jsx)(r.th,{children:"Description"}),(0,s.jsx)(r.th,{children:"vLLM metric"}),(0,s.jsx)(r.th,{children:"Triton TensorRT-LLM"}),(0,s.jsx)(r.th,{children:"trtllm-serve"}),(0,s.jsx)(r.th,{children:"SGLang"})]})}),(0,s.jsxs)(r.tbody,{children:[(0,s.jsxs)(r.tr,{children:[(0,s.jsx)(r.td,{children:"TotalQueuedRequests"}),(0,s.jsx)(r.td,{children:"Gauge"}),(0,s.jsx)(r.td,{children:"The current total number of requests in the queue."}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"vllm:num_requests_waiting"})}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"nv_trt_llm_request_metrics{request_type=waiting}"})}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"trtllm_num_requests_waiting"})}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"sglang_num_queue_reqs"})})]}),(0,s.jsxs)(r.tr,{children:[(0,s.jsx)(r.td,{children:"TotalRunningRequests"}),(0,s.jsx)(r.td,{children:"Gauge"}),(0,s.jsx)(r.td,{children:"The current total number of requests actively being served on the model server."}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"vllm:num_requests_running"})}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"nv_trt_llm_request_metrics{request_type=scheduled}"})}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"trtllm_num_requests_running"})}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"sglang_num_running_reqs"})})]}),(0,s.jsxs)(r.tr,{children:[(0,s.jsx)(r.td,{children:"KVCacheUtilization"}),(0,s.jsx)(r.td,{children:"Gauge"}),(0,s.jsx)(r.td,{children:"The current KV cache utilization in percentage."}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"vllm:kv_cache_usage_perc"})}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"nv_trt_llm_kv_cache_block_metrics{kv_cache_block_type=fraction}"})}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"trtllm_kv_cache_utilization"})}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"sglang_token_usage"})})]}),(0,s.jsxs)(r.tr,{children:[(0,s.jsx)(r.td,{children:"[Optional] BlockSize"}),(0,s.jsx)(r.td,{children:"Labeled/Gauge"}),(0,s.jsxs)(r.td,{children:["The block size in tokens to allocate memory, used by the prefix cache scorer. If this metric is not available, the BlockSize will be derived from the ",(0,s.jsx)(r.a,{href:"https://gateway-api-inference-extension.sigs.k8s.io/guides/epp-configuration/prefix-aware/#customize-the-prefix-cache-plugin",children:"prefix plugin config"}),"."]}),(0,s.jsxs)(r.td,{children:["name: ",(0,s.jsx)(r.code,{children:"vllm:cache_config_info"}),", label name: ",(0,s.jsx)(r.code,{children:"block_size"})]}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"nv_trt_llm_kv_cache_block_metrics{kv_cache_block_type=tokens_per}"})}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"trtllm_kv_cache_tokens_per_block"})}),(0,s.jsxs)(r.td,{children:["name: ",(0,s.jsx)(r.code,{children:"sglang_cache_config_info"}),", label name: ",(0,s.jsx)(r.code,{children:"page_size"})]})]}),(0,s.jsxs)(r.tr,{children:[(0,s.jsx)(r.td,{children:"[Optional] NumGPUBlocks"}),(0,s.jsx)(r.td,{children:"Labeled/Gauge"}),(0,s.jsxs)(r.td,{children:["The total number of blocks in the HBM KV cache, used by the prefix cache scorer. If this metric is not available, the NumGPUBlocks will be derived from the ",(0,s.jsx)(r.a,{href:"https://gateway-api-inference-extension.sigs.k8s.io/guides/epp-configuration/prefix-aware/#customize-the-prefix-cache-plugin",children:"prefix plugin config"}),"."]}),(0,s.jsxs)(r.td,{children:["name: ",(0,s.jsx)(r.code,{children:"vllm:cache_config_info"}),", label name: ",(0,s.jsx)(r.code,{children:"num_gpu_blocks"})]}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"nv_trt_llm_kv_cache_block_metrics{kv_cache_block_type=max}"})}),(0,s.jsx)(r.td,{children:(0,s.jsx)(r.code,{children:"trtllm_kv_cache_max_blocks"})}),(0,s.jsxs)(r.td,{children:["name: ",(0,s.jsx)(r.code,{children:"sglang_cache_config_info"}),", label name: ",(0,s.jsx)(r.code,{children:"num_pages"})]})]})]})]}),"\n",(0,s.jsx)(r.p,{children:"To correctly map metrics names, model server Pods should be labeled with the model server type they are running as demonistrated below. Pods without the engine-type label will default to vLLM metrics names."}),"\n",(0,s.jsx)(r.pre,{children:(0,s.jsx)(r.code,{className:"language-yaml",children:"metadata:\n labels:\n llm-d.ai/engine-type: vllm # other options: sglang, trtllm-serve, triton-tensorrt-llm\n\n"})}),"\n",(0,s.jsx)(r.admonition,{type:"note",children:(0,s.jsxs)(r.p,{children:[(0,s.jsxs)(r.strong,{children:["TensorRT-LLM (",(0,s.jsx)(r.code,{children:"trtllm-serve"}),") requirements."]})," Unlike vLLM/SGLang, ",(0,s.jsx)(r.code,{children:"trtllm-serve"})," exposes\nthe metrics above at ",(0,s.jsx)(r.strong,{children:(0,s.jsx)(r.code,{children:"/prometheus/metrics"})}),". The plain ",(0,s.jsx)(r.code,{children:"/metrics"})," route returns JSON\niteration-stats the EPP cannot parse, so point the EPP's metrics data source at\n",(0,s.jsx)(r.code,{children:"path: /prometheus/metrics"}),". The gauges are emitted only when the server is started with\n",(0,s.jsx)(r.strong,{children:"both"})," ",(0,s.jsx)(r.code,{children:"return_perf_metrics: true"})," ",(0,s.jsx)(r.strong,{children:"and"})," ",(0,s.jsx)(r.code,{children:"enable_iter_perf_stats: true"})," (both default\n",(0,s.jsx)(r.code,{children:"false"}),", passed via ",(0,s.jsx)(r.code,{children:"--extra_llm_api_options"}
1),"). The first mounts the Prometheus endpoint,\nand the second starts the iteration-stats loop that populates the dynamic gauges\n(",(0,s.jsx)(r.code,{children:"trtllm_num_requests_waiting"}),", ",(0,s.jsx)(r.code,{children:"trtllm_num_requests_running"}),", ",(0,s.jsx)(r.code,{children:"trtllm_kv_cache_utilization"}),").\nThey require ",(0,s.jsx)(r.strong,{children:"TensorRT-LLM v1.3.0rc12 or newer"})," (added in ",(0,s.jsx)(r.a,{href:"https://github.com/NVIDIA/TensorRT-LLM/pull/12545",children:"PR #12545"}),"). Earlier releases\n(including 1.2.1 GA) expose only request-lifecycle histograms. See the\n",(0,s.jsx)(r.a,{href:"https://github.com/llm-d/llm-d/tree/main/guides/optimized-baseline",children:"optimized-baseline TensorRT-LLM recipe"})," for a\nworking configuration."]})}),"\n",(0,s.jsx)(r.h3,{id:"lora-adapter-serving",children:"LoRA Adapter Serving"}),"\n",(0,s.jsx)(r.p,{children:"Model servers that support dynamic LoRA serving can benefit from the LoRA affinity algorithm. Note\nthe current algorithm in the reference EPP is highly biased towards vLLM's current dynamic LoRA\nimplementation."}),"\n",(0,s.jsxs)(r.p,{children:["The model servers MUST support serving a LoRA adapter specified in the ",(0,s.jsx)(r.code,{children:"model"})," argument of the\nrequest, provided the requested adapter is valid."]}),"\n",(0,s.jsx)(r.p,{children:"The model server MUST expose the following LoRA adapter metrics via the same Prometheus endpoint:"}),"\n",(0,s.jsxs)(r.ul,{children:["\n",(0,s.jsxs)(r.li,{children:["Metric name implemented in vLLM: ",(0,s.jsx)(r.code,{children:"vllm:lora_requests_info"})]}),"\n",(0,s.jsx)(r.li,{children:"Metric type: Gauge"}),"\n",(0,s.jsx)(r.li,{children:"Metric value: The last updated timestamp (so the EPP can find the latest)."}),"\n",(0,s.jsxs)(r.li,{children:["Metric labels:\n",(0,s.jsxs)(r.ul,{children:["\n",(0,s.jsxs)(r.li,{children:[(0,s.jsx)(r.code,{children:"max_lora"}),": The maximum number of adapters that can be loaded to GPU memory to serve a batch.\nRequests will be queued if the model server has reached MaxActiveAdapter and cannot load the\nrequested adapter. Example: ",(0,s.jsx)(r.code,{children:'"max_lora": "8"'}),"."]}),"\n",(0,s.jsxs)(r.li,{children:[(0,s.jsx)(r.code,{children:"running_lora_adapters"}),": A comma separated list of adapters that are currently loaded in GPU\nmemory and ready to serve requests. Example: ",(0,s.jsx)(r.code,{children:'"running_lora_adapters": "adapter1, adapter2"'})]}),"\n",(0,s.jsxs)(r.li,{children:[(0,s.jsx)(r.code,{children:"waiting_lora_adapters"}),": A comma separated list of adapters that are waiting to be served. Example: ",(0,s.jsx)(r.code,{children:'"waiting_lora_adapters": "adapter1, adapter2"'})]}),"\n"]}),"\n"]}),"\n"]}),"\n",(0,s.jsx)(r.h3,{id:"prefix-cache-reuse",children:"Prefix Cache Reuse"}),"\n",(0,s.jsxs)(r.p,{children:["The EPP supports prefix cache optimized request scheduling. To benefit from the optimal prefix aware request scheduling, model servers SHOULD support prefix cache reuse, such as the ",(0,s.jsx)(r.a,{href:"https://docs.vllm.ai/en/latest/features/automatic_prefix_caching.html",children:"vllm automatic prefix caching"})," feature."]}),"\n",(0,s.jsx)(r.h3,{id:"health-checks",children:"Health Checks"}),"\n",(0,s.jsx)(r.p,{children:"Model servers are expected to expose health endpoints that Kubernetes uses for liveness and readiness probes:"}),"\n",(0,s.jsxs)(r.ul,{children:["\n",(0,s.jsxs)(r.li,{children:[(0,s.jsx)(r.strong,{children:"Liveness"}),": ",(0,s.jsx)(r.code,{children:"GET /health"})," -- confirms the server process is alive"]}),"\n",(0,s.jsxs)(r.li,{children:[(0,s.jsx)(r.strong,{children:"Readiness"}),": ",(0,s.jsx)(r.code,{children:"GET /health"})," -- confirms the server is ready to accept requests"]}),"\n"]}),"\n",(0,s.jsx)(r.h2,{id:"further-reading",children:"Further Reading"}),"\n",(0,s.jsxs)(r.ul,{children:["\n",(0,s.jsx)(r.li,{children:(0,s.jsx)(r.a,{href:"https://docs.vllm.ai/",children:"vLLM Documentation"})}),"\n",(0,s.jsx)(r.li,{children:(0,s.jsx)(r.a,{href:"https://github.com/sgl-project/sglang",children:"SGLang Documentation"})}),"\n",(0,s.jsx)(r.li,{children:(0,s.jsx)(r.a,{href:"https://nvidia.github.io/TensorRT-LLM/",children:"TensorRT-LLM Documentation"})}),"\n",(0,s.jsxs)(r.li,{children:[(0,s.jsx)(r.a,{href:"/docs/dev/architecture/core/inferencepool",children:"InferencePool"})," -- how model servers are discovered and managed"]}),"\n",(0,s.jsxs)(r.li,{children:[(0,s.jsx)(r.a,{href:"router/epp",children:"EPP"})," -- how the router routes requests to model servers informed by model servers metrics"]}),"\n"]})]})}function h(e={}){const{wrapper:r}={...(0,i.R)(),...e.components};return r?(0,s.jsx)(r,{...e,children:(0,s.jsx)(a,{...e})}):a(e)}},28453(e,r,n){n.d(r,{R:()=>l,x:()=>c});var t=n(96540);const s={},i=t.createContext(s);function l(e){const r=t.useContext(i);return t.useMemo(function(){return"function"==typeof e?e(r):{...r,...e}},[r,e])}function c(e){let r;return r=e.disableParentContext?"function"==typeof e.components?e.components(s):e.components||s:l(e.components),t.createElement(i.Provider,{value:r},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.