PageSourceSearch

https://llm-d.ai/assets/js/b3a9c227.57e94234.js

js llm-d.ai collected 2026-10-01 10:40:34 UTC 6,232 bytes, 1 lines download raw bytes

1"use strict";(globalThis.webpackChunkllm_d_website=globalThis.webpackChunkllm_d_website||[]).push([[12590],{81597(e,t,i){i.r(t),i.d(t,{assets:()=>c,contentTitle:()=>d,default:()=>h,frontMatter:()=>l,metadata:()=>n,toc:()=>a});const n=JSON.parse('{"id":"well-lit-paths/predicted-latency","title":"Predicted Latency-Based Scheduling","description":"llm-d\'s optimized baseline guide leverages load signals and prefix-cache affinity to schedule requests, combining the signals together with heuristics.","source":"@site/versioned_docs/version-0.7/well-lit-paths/predicted-latency.md","sourceDirName":"well-lit-paths","slug":"/well-lit-paths/predicted-latency","permalink":"/docs/0.7/well-lit-paths/predicted-latency","draft":false,"unlisted":false,"tags":[],"version":"0.7","frontMatter":{},"sidebar":"docsSidebar","previous":{"title":"Optimized Baseline","permalink":"/docs/0.7/well-lit-paths/optimized-baseline"},"next":{"title":"Precise Prefix Cache Aware Routing","permalink":"/docs/0.7/well-lit-paths/precise-prefix-cache-aware"}}');var s=i(74848),r=i(28453);const l={},d="Predicted Latency-Based Scheduling",c={},a=[{value:"Deploy",id:"deploy",level:2},{value:"Architecture",id:"architecture",level:2},{value:"Further Reading",id:"further-reading",level:2}];function o(e){const t={a:"a",admonition:"admonition",code:"code",h1:"h1",h2:"h2",header:"header",img:"img",li:"li",p:"p",picture:"picture",source:"source",strong:"strong",ul:"ul",...(0,r.R)(),...e.components};return(0,s.jsxs)(s.Fragment,{children:[(0,s.jsx)(t.header,{children:(0,s.jsx)(t.h1,{id:"predicted-latency-based-scheduling",children:"Predicted Latency-Based Scheduling"})}),"\n",(0,s.jsxs)(t.p,{children:["llm-d's ",(0,s.jsx)(t.a,{href:"/docs/0.7/well-lit-paths/optimized-baseline",children:"optimized baseline guide"})," leverages load signals and prefix-cache affinity to schedule requests, combining the signals together with heuristics."]}),"\n",(0,s.jsx)(t.p,{children:"This path is for operators who want to adopt predicted latency-based scheduling - which uses an XGBoost model trained online - to make scheduling decisions. This strategy is useful when:"}),"\n",(0,s.jsxs)(t.ul,{children:["\n",(0,s.jsxs)(t.li,{children:["Your workload has ",(0,s.jsx)(t.strong,{children:"high variance in prompt and completion length"}),", and queue depth alone is a poor proxy for true load."]}),"\n",(0,s.jsxs)(t.li,{children:["Your clients can express ",(0,s.jsx)(t.strong,{children:"per-request latency SLOs"})," (interactive vs. batch) and you want the gateway to enforce them."]}),"\n",(0,s.jsxs)(t.li,{children:["Static weight tuning between cache affinity and load has become ",(0,s.jsx)(t.strong,{children:"fragile"})," as traffic shifts."]}),"\n"]}),"\n",(0,s.jsx)(t.admonition,{type:"note",children:(0,s.jsxs)(t.p,{children:["Predicted latency is not a fit when the pool is ",(0,s.jsx)(t.strong,{children:"heterogeneous"})," \u2014 mixed GPU types, model variants (e.g. prefill vs decode), or serving configurations in the same pool will produce inaccurate predictions, because the predictor assumes a single pod shape."]})}),"\n",(0,s.jsx)(t.h2,{id:"deploy",children:"Deploy"}),"\n",(0,s.jsxs)(t.p,{children:["See the ",(0,s.jsx)(t.a,{href:"https://github.com/llm-d/llm-d/tree/main/guides/predicted-latency-based-scheduling",children:"Predicted Latency guide"})," for manifests and step-by-step deployment."]}),"\n",(0,s.jsx)(t.h2,{id:"architecture",children:"Architecture"}),"\n",(0,s.jsxs)(t.p,{align:"center",children:["\n  ",(0,s.jsxs)(t.picture,{children:["\n    ",(0,s.jsx)(t.source,{media:"(prefers-color-scheme: dark)"}),"\n    ",(0,s.jsx)(t.img,{src:"/img/versioned/0.7/assets/latency-predictor.svg",alt:"Latency Predictor"}),"\n  "]}),"\n"]}),"\n",(0,s.jsx)(t.p,{children:"The setup deploys an EPP with the predicted latency sidecar containers:"}),"\n",(0,s.jsxs)(t.ul,{children:["\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.strong,{children:"Training Server"})," - trains the XGBoost model to predict TPOT and TTFT based on observed traffic"]}),"\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.strong,{children:"Prediction Servers"})," - predict TPOT and TTFT of the request based on current server state"]}),"\n"]}),"\n",(0,s.jsx)(t.p,{children:"During the standard request flow:"}),"\n",(0,s.jsxs)(t.ul,{children:["\n",(0,s.jsx)(t.li,{children:"Request arrives at the proxy, which forwards the request to the EPP"}),"\n",(0,s.jsx)(t.li,{children:"EPP queries the prediction server"}),"\n",(0,s.jsxs)(t.li,{children:["EPP (using ",(0,s.jsx)(t.code,{children:"latency-scorer"}),") selects optimal endpoint based on the prediction"]}),"\n",(0,s.jsx)(t.li,{children:"Proxy forwards request to the vLLM endpoint"}),"\n",(0,s.jsx)(t.li,{children:"vLLM endpoint processes the request, returns response to proxy"}),"\n",(0,s.jsx)(t.li,{children:"Proxy sends results to the training server, which uses samples to update the model"}),"\n"]}),"\n",(0,s.jsx)(t.h2,{id:"further-reading",children:"Further Reading"}),"\n",(0,s.jsxs)(t.ul,{children:["\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.a,{href:"/docs/0.7/architecture/advanced/latency-predictor",children:"Latency Predictor Architecture"})," \u2014 plugin pipeline, ML model, scaling characteristics, metric reference."]}),"\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.a,{href:"https://github.com/llm-d/llm-d-latency-predictor",children:"llm-d/llm-d-latency-predictor"})," \u2014 source for the training and prediction server Python code."]}),"\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.a,{href:"https://llm-d.ai/blog/predicted-latency-based-scheduling-for-llms",children:"Predicted Latency-Based Scheduling for LLMs - Blog"})," \u2014 design rationale and benchmark results."]}),"\n"]})]})}function h(e={}){const{wrapper:t}={...(0,r.R)(),...e.components};return t?(0,s.jsx)(t,{...e,children:(0,s.jsx)(o,{...e})}):o(e)}},28453(e,t,i){i.d(t,{R:()=>l,x:()=>d});var n=i(96540);const s={},r=n.createContext(s);function l(e){const t=n.useContext(r);return n.useMemo(function(){return"function"==typeof e?e(t):{...t,...e}},[t,e])}function d(e){let t;return t=e.disableParentContext?"function"==typeof e.components?e.components(s):e.components||s:l(e.components),n.createElement(r.Provider,{value:t},e.children)}}}]);

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.