PageSourceSearch

https://llm-d.ai/assets/js/77b2d815.b01f5232.js

js llm-d.ai collected 2026-10-01 10:41:44 UTC 36,484 bytes, 1 lines download raw bytes

1"use strict";(globalThis.webpackChunkllm_d_website=globalThis.webpackChunkllm_d_website||[]).push([[23327],{17838(e,t,i){i.r(t),i.d(t,{assets:()=>o,contentTitle:()=>r,default:()=>h,frontMatter:()=>l,metadata:()=>n,toc:()=>c});var n=i(89843),a=i(74848),s=i(28453);const l={title:"BLIS: Evolving llm-d at Simulation Speed",description:"BLIS is a calibrated discrete-event simulator for llm-d control-plane behavior. It helps developers evaluate routing, admission, KV cache, batching, prefill/decode placement, and capacity choices before spending time on cluster validation.",slug:"blis-evolving-llm-d-at-simulation-speed",date:"2026-06-05T09:00",authors:["merttoslali","dipanwitaguhathakurta","srinivasanparthasarathy","jingchen","nickmasluk","vishakharamani","michaelkalantar","assertantawi","fabiooliveira","carloscosta"],tags:["blog"]},r="BLIS: Evolving llm-d at Simulation Speed",o={authorsImageUrls:[void 0,void 0,void 0,void 0,void 0,void 0,void 0,void 0,void 0,void 0]},c=[{value:"The cost of policy search",id:"the-cost-of-policy-search",level:2},{value:"What is BLIS?",id:"what-is-blis",level:2},{value:"Fidelity and validation",id:"fidelity-and-validation",level:2},{value:"AI-native evolution of llm-d",id:"ai-native-evolution-of-llm-d",level:2},{value:"Case study: from latency cliffs to graceful admission control",id:"case-study-from-latency-cliffs-to-graceful-admission-control",level:3},{value:"Policy space: when to disaggregate prefill and decode",id:"policy-space-when-to-disaggregate-prefill-and-decode",level:3},{value:"Capacity planning",id:"capacity-planning",level:3}
1,{value:"Why this matters",id:"why-this-matters",level:2},{value:"Limitations",id:"limitations",level:2},{value:"What's next",id:"whats-next",level:2},{value:"Further reading",id:"further-reading",level:3},{value:"Get Involved with llm-d",id:"get-involved-with-llm-d",level:2}];function d(e){const t={a:"a",admonition:"admonition",em:"em",h2:"h2",h3:"h3",hr:"hr",img:"img",li:"li",p:"p",strong:"strong",table:"table",tbody:"tbody",td:"td",th:"th",thead:"thead",tr:"tr",ul:"ul",...(0,s.R)(),...e.components};return(0,a.jsxs)(a.Fragment,{children:[(0,a.jsx)(t.p,{children:"Deploying llm-d is not just a question of choosing a model server and adding GPUs. In a production inference deployment, operators have to choose routing policies, admission behavior, batching settings, KV-cache reuse strategies, prefill/decode placement, and autoscaling rules under concrete TTFT, ITL, throughput, and cost constraints."}),"\n",(0,a.jsx)(t.p,{children:"These choices are coupled. A routing change that improves cache locality can concentrate load. A prefill/decode threshold that helps one workload can hurt another. An admission policy that protects critical traffic can reduce total served volume. A change in any one policy can shift TTFT, inter-token latency, throughput, SLO compliance, and accelerator cost in ways that are difficult to predict analytically."}),"\n",(0,a.jsx)(t.p,{children:"The only reliable way to confirm those tradeoffs is to measure them in a GPU-backed llm-d cluster. But using cluster runs as the first step in every policy or capacity-planning experiment is too slow and expensive. BLIS provides a faster inner loop: a calibrated discrete-event simulator for distributed inference systems like llm-d. Developers can evaluate candidate policies and deployment configurations locally, then reserve cluster validation for the candidates most likely to matter."}),"\n",(0,a.jsx)(t.admonition,{title:"Blog key takeaways",type:"tip",children:(0,a.jsxs)(t.ul,{children:["\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"BLIS is a discrete-event simulator:"})," \u2014 it models admission, routing, scheduling, KV cache, batching, and prefill/decode placement without loading model weights or occupying GPUs."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Calibrated fidelity:"}
1)," Median 7\u20139% error on end-to-end and inter-token latency across 36 validation experiments spanning 8B\u2013141B parameter models, H100/A100/L40S GPUs, and diverse workloads. Approximately 200\xd7 faster than equivalent cluster runs."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Admission control case study:"})," An AI-native policy-search loop using BLIS discovered a probabilistic admission controller that reduced critical-tier TTFT p90 by up to 97% and end-to-end latency by up to 50%, validated on a real llm-d cluster."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Capacity planning:"})," BLIS evaluates hundreds of deployment configurations in minutes, producing ranked Pareto-optimal candidates before any GPU time is spent."]}),"\n"]})}),"\n",(0,a.jsx)(t.h2,{id:"the-cost-of-policy-search",children:"The cost of policy search"}),"\n",(0,a.jsx)(t.p,{children:"The case for simulation is strongest when deployment choices form a large search space. In llm-d, that search space includes parallelism strategy, replica topology, routing policy, admission behavior, batching configuration, KV-cache reuse, and prefill/decode placement. Each choice affects the others, so the best configuration is usually workload- and SLO-dependent rather than universal."}),"\n",(0,a.jsx)(t.p,{children:"The need is especially clear in production-style inference. A useful evaluation often requires realistic request distributions, multi-instance topologies, enough offered load to expose saturation behavior, and repeated runs across policy or configuration variants. These evaluations are subtle because prefill and decode stress the system differently, batching can improve throughput while shifting latency, and disaggregation only helps when transfer cost, queue state, and TTFT/ITL targets line up. BLIS changes the economics of this search:"}),"\n",(0,a.jsxs)(t.table,{children:[(0,a.jsx)(t.thead,{children:(0,a.jsxs)(t.tr,{children:[(0,a.jsx)(t.th,{style:{textAlign:"left"}}),(0,a.jsx)(t.th,{style:{textAlign:"left"},children:"BLIS"}),(0,a.jsx)(t.th,{style:{textAlign:"left"},children:"GPU cluster"})]})}),(0,a.jsxs)(t.tbody,{children:[(0,a.jsxs)(t.tr,{children:[(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Wall-clock time per config"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"~seconds"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"~hours"})]}),(0,a.jsxs)(t.tr,{children:[(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Hardware required"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"CPU (local)"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Multi-GPU (e.g. 4\u201316\xd7 H100)"})]}),(0,a.jsxs)(t.tr,{children:[(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Cost per config"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Negligible"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"GPU-hours at cluster rates"})]}),(0,a.jsxs)(t.tr,{children:[(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Configs evaluated per hour"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Hundreds"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Single digits"})]}),(0,a.jsxs)(t.tr,{children:[(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Deterministic replay"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Yes"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"No (system jitter, variance)"})]})]})]}),"\n",(0,a.jsx)(t.p,{children:"Running this search directly on GPU-backed clusters turns policy development into a scarce-resource scheduling problem. BLIS changes the order of operations: broad exploration happens through local simulation, and cluster validation is used later for the small set of policies or configurations that simulation identifies as worth the cluster time."}),"\n",(0,a.jsx)(t.h2,{id:"what-is-blis",children:"What is BLIS?"}),"\n",(0,a.jsx)(t.p,{children:"BLIS models the parts of distributed LLM serving that determine system-level behavior: request arrival, admission, routing, queueing, chunked prefill, continuous batching, decode scheduling, KV-cache allocation and reuse, prefill/decode transfer costs, and multi-instance placement. It does not load model weights or execute tensor kernels. Instead, it advances the request lifecycle through a discrete-event simulation driven by performance models fit to real measurements."}),"\n",(0,a.jsx)("div",{style:{margin:"20px 0"},children:(0,a.jsx)("img",{src:"/img/blogs/blis-evolving-llm-d-at-simulation-speed/twin-diagram.svg",alt:"Real llm-d vs BLIS architecture",style:{width:"100%",height:"auto"}})}),"\n",(0,a.jsx)("small",{children:(0,a.jsxs)(t.em,{children:[(0,a.jsx)(t.strong,{children:"FIGURE 1"}),": BLIS architecture. The simulator consumes the same inputs operators reason about when deploying llm-d \u2014 workload traces, model profiles, topology, policies, and vLLM configuration \u2014 and produces the metrics that decide whether a policy is viable. Internally, calibrated models replace GPU execution for each stage of the request lifecycle."]})}),"\n",(0,a.jsx)("br",{}),"\n",(0,a.jsx)("br",{}),"\n",(0,a.jsx)(t.p,{children:"BLIS models each stage of the request lifecycle \u2014 admission, routing, queueing, chunked prefill, continuous batching, decode scheduling, KV-cache allocation, and P/D transfer \u2014 using performance models fit to real GPU measurements. It produces TTFT, inter-token latency, end-to-e
1nd latency, throughput, queue depth, SLO violations, shed traffic, and policy rankings without occupying cluster accelerators."}),"\n",(0,a.jsx)(t.p,{children:"That makes BLIS useful in two different workflows:"}),"\n",(0,a.jsxs)(t.ul,{children:["\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Policy development:"})," compare routing, admission, batching, P/D placement, and autoscaling behavior before implementing or validating the best candidates in production code."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Capacity planning:"})," sweep hardware budgets and vLLM/llm-d configuration choices to identify the smallest set of real deployments worth benchmarking."]}),"\n"]}),"\n",(0,a.jsx)(t.h2,{id:"fidelity-and-validation",children:"Fidelity and validation"}),"\n",(0,a.jsx)(t.p,{children:"BLIS is not intended to replace cluster validation. Its job is to make exploration cheap enough that developers can search a larger space before scheduling cluster runs. For that to work, BLIS must be accurate enough to identify promising candidates and preserve the relative ranking of alternatives."}),"\n",(0,a.jsx)(t.p,{children:"In current validation, BLIS shows median 7\u20139% error on end-to-e
1nd and inter-token latency relative to cluster runs. Equivalent cluster experiments take roughly 200\xd7 longer to run."}),"\n",(0,a.jsxs)(t.admonition,{title:"Fidelity validation scope",type:"info",children:[(0,a.jsxs)(t.p,{children:["The validation set spans ",(0,a.jsx)(t.strong,{children:"36 experiments"})," across:"]}),(0,a.jsxs)(t.ul,{children:["\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Models:"})," Dense and MoE architectures from 8B to 141B parameters (Llama, Mixtral, Qwen families)"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Workloads:"})," Chat, code generation, and long-output reasoning"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"GPUs:"})," NVIDIA H100, A100, and L40S"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Configuration sweeps:"})," Tensor parallelism and chunk size variations"]}),"\n"]})]}),"\n",(0,a.jsx)(t.p,{children:"The most important fidelity question depends on the use case. For capacity planning, absolute latency error matters because SLO boundaries determine feasible configurations. For policy search, rank fidelity is often more important: if policy A beats policy B in BLIS across representative workloads, cluster validation should confirm the same ordering often enough to make simulation a reliable filter."}),"\n",(0,a.jsx)(t.h2,{id:"ai-native-evolution-of-llm-d",children:"AI-native evolution of llm-d"}),"\n",(0,a.jsx)(t.p,{children:"AI-native evolution means using agents to propose, test, and refine policies or new algorithms, while reserving cluster validation for the most promising candidates. BLIS is the fast inner loop; a GPU-backed llm-d deployment is the validation outer loop. Developers can connect BLIS to any policy-search workflow \u2014 a human-driven sweep, a custom optimizer, or an agentic system \u2014 and explore routing, admission, batching, P/D placement, cache behavior, and capacity choices in simulation before validating the strongest candidates on a real cluster. Measurements from cluster runs feed back into calibration, making each round of simulation more accurate. The admission-control case study below shows this pipeline end to end."}),"\n",(0,a.jsx)(t.h3,{id:"case-study-from-latency-cliffs-to-graceful-admission-control",children:"Case study: from latency cliffs to graceful admission control"}),"\n",(0,a.jsx)(t.p,{children:"Admission control is a good example of why a simulator is useful. Under overload, default llm-d admission behavior can act like a cliff: requests are admitted until saturation is reached, then sheddable traffic is rejected hard. By the time the threshold fires, queues may already be deep enough to affect protected traffic."}),"\n",(0,a.jsx)(t.p,{children:"Figure 2 shows how the AI-native pipeline was applied to this problem. BLIS evaluated many candidate admission policies across workload traces, ranked them by SLO compliance, and passed the strongest candidates to cluster validation. Cluster measurements then fed back to calibrate the next round of simulation."}),"\n",(0,a.jsx)("div",{style:{margin:"20px 0"},children:(0,a.jsx)("img",{src:"/img/blogs/blis-evolving-llm-d-at-simulation-speed/ai-native-loop.svg",alt:"AI-native pipeline: workload traces through BLIS evaluation, ranked candidates, cluster validation, to upstream contribution",style:{width:"90%",height:"auto"}})}),"\n",(0,a.jsx)("small",{children:(0,a.jsxs)(t.em,{children:[(0,a.jsx)(t.strong,{children:"FIGURE 2"}),": The AI-native pipeline applied to admission control. Workload traces and policy candidates flow through BLIS batch evaluation, producing a ranked shortlist filtered by SLO. The strongest candidates are validated on a real llm-d cluster. Validated policies become upstream contributions. Cluster measurements feed back to calibrate BLIS for the next round."]})}),"\n",(0,a.jsx)("br",{}),"\n",(0,a.jsx)("br",{}),"\n",(0,a.jsx)(t.p,{children:"The pipeline narrowed a large policy space to a single winner: a probabilistic admitter that sheds low-priority traffic gradually as saturation rises, protecting critical traffic before the system reaches the cliff. Cluster validation on Qwen3-14B served by vLLM on 4\xd7 NVIDIA H100-SXM-80GB GPUs, routed through llm-d, confirmed the improvement:"}),"\n",(0,a.jsxs)(t.table,{children:[(0,a.jsx)(t.thead,{children:(0,a.jsxs)(t.tr,{children:[(0,a.jsx)(t.th,{style:{textAlign:"left"},children:"Metric (critical tier, overloaded)"}),(0,a.jsx)(t.th,{style:{textAlign:"left"},children:"Default hard-shed"}),(0,a.jsx)(t.th,{style:{textAlign:"left"},children:"Probabilistic admitter"}),(0,a.jsx)(t.th,{style:{textAlign:"left"},children:"Improvement"})]})}),(0,a.jsxs)(t.tbody,{children:[(0,a.jsxs)(t.tr,{children:[(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"TTFT p90"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Latency cliff under overload"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Smooth degradation"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Up to 97% reduction"})]}),(0,a.jsxs)(t.tr,{children:[(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"End-to-end latency"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Queues build before shed fires"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Early shed prevents buildup"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Up to 50% reduction"})]}),(0,a.jsxs)(t.tr,{children:[(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Shed behavior"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Abrupt rejection at threshold"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Gradual shedding as saturation rises"}),(0,a.jsx)(t.td,{style:{textAlign:"left"},children:"Graceful"})]})]})]}),"\n",(0,a.jsx)("div",{style:{textAlign:"center",margin:"20px 0"},children:(0,a.jsx)("img",{src:"/img/blogs/blis-evolving-llm-d-at-simulation-speed/admission-before-after.svg",alt:"Admission control before and after: latency cliff vs smooth degradation",style:{width:"100%",height:"auto"}})}),"\n",(0,a.jsx)("small",{children:(0,a.jsxs)(t.em,{children:[(0,a.jsx)(t.strong,{children:"FIGURE 3"}),": Critical-tier TTFT p90 under increasing load. The default hard-shed policy (red) holds steady until saturation, then degrades sharply past the SLO. The probabilistic admitter (green) sheds low-priority traffic gradually, keeping critical-tier latency closer to the SLO target through overload."]})}),"\n",(0,a.jsx)("br",{}),"\n",(0,a.jsx)("br",{}),"\n",(0,a.jsxs)(t.p,{children:["The important point for llm-d is not only the specific admission algorithm. It is the workflow: simulation narrowed a large policy space, cluster validation confirmed the strongest candidate, and the result became a concrete llm-d-router contribution. For the full discovery process, algorithm details, and benchmark matrix, see the ",(0,a.jsx)(t.a,{href:"https://ai-native-systems-research.github.io/ai-native-systems-research/blog/2026/05/13/from-simulation-to-production-how-an-ai-native-pipeline-discovered-a-better-admission-controller-for-llm-d/",children:"admission-controller case study"}),"."]}),"\n",(0,a.jsx)(t.h3,{id:"policy-space-when-to-disaggregate-prefill-and-decode",children:"Policy space
1: when to disaggregate prefill and decode"}),"\n",(0,a.jsx)(t.p,{children:"llm-d's current P/D decider uses a fixed prefix-cache threshold: disaggregate any request with more than N=16 uncached tokens. This is cache-aware but queue-blind \u2014 it disaggregates at the same rate regardless of whether the prefill pool is idle or saturated."}),"\n",(0,a.jsx)(t.p,{children:"That threshold is simple and often useful, but it leaves performance on the table when queue state matters more than uncached-token count. A short uncached prompt may not justify KV-transfer overhead even if it crosses the threshold. A long uncached prompt may benefit from disaggregation only when the prefill pool has enough spare capacity to absorb it. The right decision depends on request shape, cache state, queue depth, transfer cost, and SLO pressure."}),"\n",(0,a.jsx)(t.p,{children:"BLIS lets us evaluate a wider policy space: always-local, always-disaggregate, stationary randomized, fixed threshold, and a dynamic policy we call Empirical Drift-Plus-Penalty (EDPP), derived from Lyapunov optimization. EDPP routes each request using two signals: the relative queue depths of the decode and prefill pools at decision time, and a virtual TTFT queue that accumulates deficit whenever a disaggregated request misses an operator-specified TTFT SLO. When the decode pool is backlogged and the prefill pool has spare capacity, EDPP disaggregates. When prior disaggregation decisions push TTFT past the SLO, the virtual queue grows and suppresses future disaggregation until TTFT recovers. No hardware constants are required; the operator specifies goals, including an ITL target and TTFT SLO, and the policy adapts."}),"\n",(0,a.jsx)(t.p,{children:"We evaluated on a 1P+3D topology (1 prefill + 3 decode instances) using BLIS's trained-physics latency model configured for Meta Llama 3.1-70B-Instruct on NVIDIA H100 (TP=4 per instance, 16 H100s total), with workloads from the inference-perf catalog:"}),"\n",(0,a.jsxs)(t.ul,{children:["\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Interactive chat:"})," 5K-token prefix, approximately 50 uncached tokens per turn, 4 turns per session. At N=16, the threshold decider disaggregates nearly every turn. Because uncached inputs are short, the KV-transfer round trip adds overhead with little throughput benefit. EDPP disaggregates only when decode backlog makes the transfer worthwhile, reducing mean TTFT by 2-3x at moderate-to-high load in BLIS."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Code generation:"})," 30K-token prefix, approximately 1,500 uncached tokens per turn, 15 turns per session. At N=16, the threshold fires for 100% of requests. This can saturate the prefill pool and inflate TTFT by up to 20x relative to always-local. EDPP's SLO-feedback loop suppresses disaggregation when TTFT exceeds the target, stabilizing the disaggregation fraction near 50%."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Code generation:"})," 30K-token prefix, approximately 1,500 uncached tokens per turn, 15 turns per session. At N=16, the threshold fires for 100% of requests, routing all of them to the prefill pool. This can saturate prefill and inflate TTFT by up to 20\xd7 relative to always-local. EDPP avoids this: when disaggregated requests start missing the TTFT SLO, its feedback loop reduces the disaggregation rate \u2014 in this workload, settling near 50% \u2014 so the prefill pool stays below saturation and TTFT remains bounded.\nThe plot below shows one slice of this policy comparison for the interactive-chat workload."]}),"\n"]}),"\n",(0,a.jsx)(t.p,{children:(0,a.jsx)(t.img,{alt:"PD Decider: Threshold=16 vs Empirical Drift Plus Penalty",src:i(92738).A+"",width:"1050",height:"750"})}),"\n",(0,a.jsx)("small",{children:(0,a.jsxs)(t.em,{children:[(0,a.jsx)(t.strong,{children:"FIGURE 4"}),": TTFT comparison for interactive-chat workload on a 1P+3D topology (Llama 3.1-70B, 16\xd7 H100). The fixed threshold (N=16) disaggregates nearly every turn, adding KV-transfer overhead. EDPP disaggregates selectively based on queue state and SLO feedback."]})}),"\n",(0,a.jsx)("br",{}),"\n",(0,a.jsx)("br",{}),"\n",(0,a.jsx)(t.p,{children:"The point is not that BLIS eliminates the need for cluster validation. It makes this kind of policy search practical enough that cluster runs can be reserved for the small number of policies that survive broad simulated evaluation."}),"\n",(0,a.jsx)(t.h3,{id:"capacity-planning",children:"Capacity planning"}),"\n",(0,a.jsx)(t.p,{children:"The same simulation loop applies to deployment planning. Before committing cluster time, operators need to know which configurations can plausibly meet a workload's SLO:"}),"\n",(0,a.jsxs)(t.ul,{children:["\n",(0,a.jsx)(t.li,{children:"How many GPUs? Which GPU type?"}
1),"\n",(0,a.jsx)(t.li,{children:"What configurations meet the SLO?"}),"\n",(0,a.jsx)(t.li,{children:"Which router knobs \u2014 scorer weights, prefix-cache priority, load-balance settings?"}),"\n",(0,a.jsx)(t.li,{children:"Which vLLM knobs \u2014 tensor parallelism, chunk size, batch limits?"}),"\n"]}),"\n",(0,a.jsx)(t.p,{children:"Without simulation, each candidate can become a cluster experiment. With BLIS, configuration search becomes a local CPU-based sweep that produces a ranked set of viable options before validation time is spent on the cluster."}),"\n",(0,a.jsx)(t.p,{children:(0,a.jsx)(t.img,{alt:"BLIS Config Search: Pareto Frontier for Llama-3.1-70B on H100",src:i(52295).A+"",width:"1999",height:"1263"})}),"\n",(0,a.jsx)("small",{children:(0,a.jsxs)(t.em,{children:[(0,a.jsx)(t.strong,{children:"FIGURE 5"}),": Pareto frontier from a BLIS configuration sweep. Each dot is a candidate deployment (varying TP, replica count, batch limits, and cache settings). The dashed line marks a TTFT SLO boundary; starred points are the highest-throughput feasible configurations at each GPU budget tier."]})}),"\n",(0,a.jsx)("br",{}),"\n",(0,a.jsx)("br",{}),"\n",(0,a.jsx)(t.p,{children:"In this kind of view, each dot is a BLIS-evaluated configuration. The x-axis captures latency, the y-axis captures sustainable throughput, and the SLO boundary separates feasible from infeasible candidates. The Pareto frontier identifies the highest-throughput configurations at each feasible latency/cost point, while annotations explain the concrete deployment choices behind selected budget tiers."}),"\n",(0,a.jsx)(t.p,{children:"This is the role BLIS can play in llm-d capacity planning: not to hand operators a single answer without validation, but to reduce an enormous search space to a short list of explainable candidates."}),"\n",(0,a.jsxs)(t.p,{children:[(0,a.jsx)(t.a,{href:"https://github.com/llm-d-incubation/llm-d-planner",children:"llm-d-planner"}),", the deployment-recommendation tool for llm-d, is planned to consume BLIS output to power exactly this kind of sizing and policy advice."]}),"\n",(0,a.jsx)(t.hr,{}),"\n",(0,a.jsx)(t.h2,{id:"why-this-matters",children:"Why this matters"}),"\n",(0,a.jsx)(t.p,{children:"llm-d's mission is to make advanced inference optimizations practical for production Kubernetes deployments. That requires more than peak benchmark results. It requires a way to reason about policy interactions, workload sensitivity, and cost before users spend cluster time."}),"\n",(0,a.jsx)(t.p,{children:"BLIS gives llm-d a fast, deterministic inner loop for that work. Developers can test routing changes, admission policies, batching behavior, prefill/decode decisions, and capacity assumptions in simulation, then reserve cluster runs for the candidates worth validating."}),"\n",(0,a.jsx)(t.p,{children:"That is the practical shift: cluster validation remains mandatory, but broad cluster exploration becomes targeted. For a project like llm-d, where the control plane is itself a source of performance advantage, that faster inner loop is infrastructure for continued evolution."}),"\n",(0,a.jsx)(t.hr,{}),"\n",(0,a.jsx)(t.h2,{id:"limitations",children:"Limitations"}),"\n",(0,a.jsx)(t.p,{children:"The following areas highlight current limitations of BLIS:"}),"\n",(0,a.jsxs)(t.ul,{children:["\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Network effects:"})," BLIS models tensor-parallel and data-parallel communication overhead from profiling data, but does not explicitly model the network itself. Real network behavior varies with hardware topology, region, and is subject to jitter \u2014 none of which are captured."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Platform drift:"})," The simulator mirrors vLLM and llm-d behavior at a point in time. As the real stack evolves, the simulator must be updated to stay accurate."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Selective fidelity:"})," At any point in time, BLIS is not expected to model everything in llm-d and vLLM, but only the most load-bearing aspects of the real system. We focus on a few aspects at a time to improve algorithms in the stack, and those are the aspects prioritized for development in the simulator \u2014 on an as-needed-for-policy-evolution basis."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Saturated regimes:"})," BLIS's performance model is not calibrated for deeply saturated conditions. This is expected \u2014 in heavily overloaded systems, small perturbations in arrival or service times cause disproportionate queueing effects, making precise prediction impractical for any simulator. The practical goal is to identify policies that avoid saturation, not to predict behavior within it."]}),"\n"]}),"\n",(0,a.jsx)(t.hr,{}),"\n",(0,a.jsx)(t.h2,{id:"whats-next",children:"What's next"}),"\n",(0,a.jsx)(t.p,{children:"BLIS is under active development. Key directions include:"}),"\n",(0,a.jsxs)(t.ul,{children:["\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Broader llm-d coverage:"})," Extending the simulator to track new llm-d control-plane features as they land \u2014 including autoscaling policies, multi-model routing, and evolving P/D placement strategies."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Calibration and fidelity:"})," Expanding the validation set to cover new GPU families, larger topologies, and additional workload patterns. Improving performance-model accuracy in near-saturation regimes."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Integration with llm-d-planner:"})," Connecting BLIS output to ",(0,a.jsx)(t.a,{href:"https://github.com/llm-d-incubation/llm-d-planner",children:"llm-d-planner"})," to provide deployment sizing and policy recommendations backed by simulation evidence."]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Community contribution:"})," The upcoming BLIS proposal for llm-d will outline the path for BLIS to become a community-maintained component of the llm-d ecosystem."]}),"\n"]}),"\n",(0,a.jsx)(t.h3,{id:"further-reading",children:"Further reading"}),"\n",(0,a.jsxs)(t.ul,{children:["\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"BLIS project:"})," ",(0,a.jsx)(t.a,{href:"https://inference-sim.github.io/inference-sim/latest/",children:"inference-sim.github.io/inference-sim"})]}),"\n",(0,a.jsx)(t.li,{children:(0,a.jsx)(t.a,{href:"https://inference-sim.github.io/inference-sim/latest/blog/2026/03/05/why-simulate-before-you-scale/",children:"Why simulate before you scale"})}),"\n",(0,a.jsx)(t.li,{children:(0,a.jsx)(t.a,{href:"https://medium.com/modeling-distributed-inference/the-physics-of-high-fidelity-distributed-inference-platform-simulation-28fe27b59da2",children:"The physics of high-fidelity distributed inference platform simulation"})}),"\n",(0,a.jsx)(t.li,{children:(0,a.jsx)(t.a,{href:"https://ai-native-systems-research.github.io/ai-native-systems-research/blog/2026/05/13/from-simulation-to-production-how-an-ai-native-pipeline-discovered-a-better-admission-controller-for-llm-d/",children:"From simulation to production: the admission controller case study"})}),"\n"]}),"\n",(0,a.jsx)(t.hr,{}),"\n",(0,a.jsx)(t.h2,{id:"get-involved-with-llm-d",children:"Get Involved with llm-d"}),"\n",(0,a.jsx)(t.p,{children:"The llm-d project thrives on community contributions, and there are many ways to get involved:"}),"\n",(0,a.jsxs)(t.ul,{children:["\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Explore the code"})," \u2192 Browse our ",(0,a.jsx)(t.a,{href:"https://github.com/llm-d",children:"GitHub organization"})," and dig into the projects powering this stack"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Join our Slack"})," \u2192 ",(0,a.jsx)(t.a,{href:"/slack",children:"Get your invite"})," and connect with maintainers and contributors"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Attend community calls"})," \u2192 All meetings are open! Add our ",(0,a.jsx)(t.a,{href:"https://red.ht/llm-d-public-calendar",children:"public calendar"}
1)," and join the conversation"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Follow project updates"})," \u2192 Stay current on ",(0,a.jsx)(t.a,{href:"https://twitter.com/_llm_d_",children:"Twitter/X"}),", ",(0,a.jsx)(t.a,{href:"https://bsky.app/profile/llm-d.ai",children:"Bluesky"}),", and ",(0,a.jsx)(t.a,{href:"https://www.linkedin.com/company/llm-d",children:"LinkedIn"})]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Watch demos and recordings"})," \u2192 Subscribe to the ",(0,a.jsx)(t.a,{href:"https://www.youtube.com/@llm-d-project",children:"llm-d YouTube channel"})," for community call recordings and feature walkthroughs"]}),"\n",(0,a.jsxs)(t.li,{children:[(0,a.jsx)(t.strong,{children:"Read the docs"})," \u2192 Visit our ",(0,a.jsx)(t.a,{href:"/community",children:"community page"})," to find SIGs, contribution guides, and upcoming events"]}),"\n"]})]})}function h(e={}){const{wrapper:t}={...(0,s.R)(),...e.components};return t?(0,a.jsx)(t,{...e,children:(0,a.jsx)(d,{...e})}):d(e)}},52295(e,t,i){i.d(t,{A:()=>n});const n=i.p+"assets/images/pareto-frontier-637beb34407eaabe806dfccca7dc6533.png"},92738(e,t,i){i.d(t,{A:()=>n});const n=i.p+"assets/images/pd-decider-ttft-7e4cc9c8192acdd518666ffc2048f115.png"},28453(e,t,i){i.d(t,{R:()=>l,x:()=>r});var n=i(96540);const a={},s=n.createContext(a);function l(e){const t=n.useContext(s);return n.useMemo(function(){return"function"==typeof e?e(t):{...t,...e}},[t,e])}function r(e){let t;return t=e.disableParentContext?"function"==typeof e.components?e.components(a):e.components||a:l(e.components),n.createElement(s.Provider,{value:t},e.children)}},89843(e){e.exports=JSON.parse('{"permalink":"/blog/blis-evolving-llm-d-at-simulation-speed","editUrl":"https://github.com/llm-d/llm-d/edit/main/website/blog/2026-06-05_blis-evolving-llm-d-at-simulation-speed.mdx","source":"@site/blog/2026-06-05_blis-evolving-llm-d-at-simulation-speed.mdx","title":"BLIS: Evolving llm-d at Simulation Speed","description":"BLIS is a calibrated discrete-event simulator for llm-d control-plane behavior. It helps developers evaluate routing, admission, KV cache, batching, prefill/decode placement, and capacity choices before spending time on cluster validation.","date":"2026-06-05T09:00:00.000Z","tags":[{"inline":false,"label":"blog posts","permalink":"/blog/tags/blog","description":"everyday blog posts"}],"readingTime":15.9,"hasTruncateMarker":true,"authors":[{"name":"Mert Toslali","title":"Research Scientist, IBM","email":"[email protected]","socials":{"github":"https://github.com/mtoslalibu"},"imageURL":"https://github.com/mtoslalibu.png","key":"merttoslali","page":null},{"name":"Dipanwita Guhathakurta","title":"Software Engineer, IBM","email":"[email protected]","socials":{"github":"https://github.com/susiejojo"},"imageURL":"https://github.com/susiejojo.png","key":"dipanwitaguhathakurta","page":null},{"name":"Srinivasan Parthasarathy","title":"Principal Research Scientist, IBM","email":"[email protected]","socials":{"github":"https://github.com/sriumcp"},"imageURL":"https://github.com/sriumcp.png","key":"srinivasanparthasarathy","page":null},{"name":"Jing Chen","title":"Software Engineer, IBM","email":"[email protected]","socials":{"github":"https://github.com/jgchn"},"imageURL":"https://github.com/jgchn.png","key":"jingchen","page":null},{"name":"Nick Masluk","title":"Research Scientist, IBM","email":"[email protected]","socials":{"github":"https://github.com/namasl"},"imageURL":"https://github.com/namasl.png","key":"nickmasluk","page":null},{"name":"Vishakha Ramani","title":"Research Scientist, IBM","email":"[email protected]","socials":{"github":"https://github.com/vishakha-ramani"},"imageURL":"https://github.com/vishakha-ramani.png","key":"vishakharamani","page":null},{"name":"Michael Kalantar","title":"Software Engineer, IBM","email":"[email protected]","socials":{"github":"https://github.com/kalantar"},"imageURL":"https://github.com/kalantar.png","key":"michaelkalantar","page":null},{"name":"Asser Tantawi","title":"Research Scientist, IBM","email":"[email protected]","socials":{"github":"https://github.com/atantawi"},"imageURL":"https://github.com/atantawi.png","key":"assertantawi","page":null},{"name":"Fabio Oliveira","title":"Senior Research Manager, IBM","email":"[email protected]","socials":{"github":"https://github.com/fabolive"},"imageURL":"https://github.com/fabolive.png","key":"fabiooliveira","page":null},{"name":"Carlos Costa","title":"Distinguished Engineer, IBM","email":"[email protected]","url":"https://github.com/chcost","socials":{"linkedin":"https://www.linkedin.com/in/carlos-costa-9b9b1a1/","github":"https://github.com/chcost"},"imageURL":"https://github.com/chcost.png","key":"carloscosta","page":null}],"frontMatter":{"title":"BLIS: Evolving llm-d at Simulation Speed","description":"BLIS is a calibrated discrete-event simulator for llm-d control-plane behavior. It helps developers evaluate routing, admission, KV cache, batching, prefill/decode placement, and capacity choices before spending time on cluster validation.","slug":"blis-evolving-llm-d-at-simulation-speed","date":"2026-06-05T09:00","authors":["merttoslali","dipanwitaguhathakurta","srinivasanparthasarathy","jingchen","nickmasluk","vishakharamani","michaelkalantar","assertantawi","fabiooliveira","carloscosta"],"tags":["blog"]},"unlisted":false,"prevItem":{"title":"Heterogeneous inference serving across three GPU vend
1ors with llm-d","permalink":"/blog/heterogeneous-inference-3-vendor-sovereign-cluster"},"nextItem":{"title":"No Kubernetes? No Problem: llm-d Now Runs Anywhere","permalink":"/blog/running-llm-d-without-kubernetes"}}')}}]);

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.