1"use strict";(globalThis.webpackChunkllm_d_website=globalThis.webpackChunkllm_d_website||[]).push([[21997],{90424(e,r,n){n.r(r),n.d(r,{assets:()=>l,contentTitle:()=>a,default:()=>p,frontMatter:()=>t,metadata:()=>s,toc:()=>c});const s=JSON.parse('{"id":"operations/README","title":"Operational Excellence","description":"Operational Excellence guidelines focus on overarching Day-2 site reliability engineering, cluster-wide telemetry frameworks, and safe lifecycle rollout strategies for generative AI inference deployments.","source":"@site/versioned_docs/version-0.9/operations/README.md","sourceDirName":"operations","slug":"/operations/","permalink":"/docs/operations/","draft":false,"unlisted":false,"tags":[],"version":"0.9","frontMatter":{},"sidebar":"docsSidebar","previous":{"title":"Inference Payload Processing","permalink":"/docs/architecture/advanced/inference-payload-processing/"},"next":{"title":"Observability","permalink":"/docs/operations/observability/"}}');var i=n(74848),o=n(28453);const t={},a="Operational Excellence",l={},c=[{value:"Cluster Observability",id:"cluster-observability",level:3},{value:"Disaggregated Serving Operations",id:"disaggregated-serving-operations",level:3},{value:"Zero-Downtime Rollouts",id:"zero-downtime-rollouts",level:3},{value:"Model-Aware Readiness Probes",id:"model-aware-readiness-probes",level:3},{value:"Serve External APIs",id:"serve-external-apis",level:3},{value:"Router Operations",id:"router-operations",level:3},{value:"Async Processor Operations",id:"async-processor-operations",level:3}];function d(e){const r={a:"a",h1:"h1",h3:"h3",header:"header",p:"p",...(0,o.R)(),...e.components};return(0,i.jsxs)(i.Fragment,{children:[(0,i.jsx)(r.header,{children:(0,i.jsx)(r.h1,{id:"operational-excellence",children:"Operational Excellence"})}),"\n",(0,i.jsx)(r.p,{children:"Operational Excellence guidelines focus on overarching Day-2 site reliability engineering, cluster-wide telemetry frameworks, and safe lifecycle rollout strategies for generative AI inference deployments."}),"\n",(0,i.jsxs)(r.p,{children:["While ",(0,i.jsx)(r.a,{href:"/docs/well-lit-paths/",children:"well-lit path guides"})," teach how to configure llm-d's native intelligent routing algorithms and inference optimizations, this top-level section governs enterprise cluster observability, alerting, and zero-downtime model updates."]}),"\n",(0,i.jsx)(r.h3,{id:"cluster-observability",children:(0,i.jsx)(r.a,{href:"/docs/operations/observability/",children:"Cluster Observability"})}),"\n",(0,i.jsx)(r.p,{children:"End-to-end telemetry setup, OpenTelemetry tracing, standard Prometheus metrics, PromQL dashboards, and monitoring architectures."}),"\n",(0,i.jsx)(r.h3,{id:"disaggregated-serving-operations",children:(0,i.jsx)(r.a,{href:"/docs/operations/disaggregation/",children:"Disaggregated Serving Operations"})}),"\n",(0,i.jsx)(r.p,{children:"Operational considerations and engine-specific guides (vLLM and SGLang) for dynamic connections, request cancellation, fault tolerance, and safe rollouts."}),"\n",(0,i.jsx)(r.h3,{id:"zero-downtime-rollouts",children:(0,i.jsx)(r.a,{href:"/docs/operations/rollouts/",children:"Zero-Downtime Rollouts"})}),"\n",(0,i.jsx)(r.p,{children:"Production rollout strategies including Blue-Green updates and live LoRA adapter hot-swapping without dropping active client traffic."}),"\n",(0,i.jsx)(r.h3,{id:"model-aware-readiness-probes",children:(0,i.jsx)(r.a,{href:"/docs/operations/readiness-probes",children:"Model-Aware Readiness Probes"})}),"\n",(0,i.jsx)(r.p,{children:"Kubernetes HTTP probe configurations using vLLM API endpoints to ensure pods are only marked Ready when models are fully loaded."}),"\n",(0,i.jsx)(r.h3,{id:"serve-external-apis",children:(0,i.jsx)(r.a,{href:"/docs/operations/serve-external-apis/",children:"Serve External APIs"})}),"\n",(0,i.jsx)(r.p,{children:"Deploy LiteLLM Proxy or Kong AI Gateway to route traffic seamlessly between self-hosted llm-d inference stacks and external cloud provider LLM APIs."}),"\n",(0,i.jsx)(r.h3,{id:"router-operations",children:(0,i.jsx)(r.a,{href:"/docs/operations/router",children:"Router Operations"})}),"\n",(0,i.jsx)(r.p,{children:"Operational best practices, high availability scaling modes, standalone proxy architectures, and container resource sizing for llm-d Router deployments."}),"\n",(0,i.jsx)(r.h3,{id:"async-processor-operations",children:(0,i.jsx)(r.a,{href:"/docs/operations/async-processor",children:"Async Processor Operations"})}),"\n",(0,i.jsx)(r.p,{children:"Throughput modeling, concurrency sizing (backed by a measured sweep), container resource sizing, and horizontal scaling for the Async Processor batch-dispatch agent."})]})}function p(e={}){const{wrapper:r}={...(0,o.R)(),...e.components};return r?(0,i.jsx)(r,{...e,children:(0,i.jsx)(d,{...e})}):d(e)}},28453(e,r,n){n.d(r,{R:()=>t,x:()=>a});var s=n(96540);const i={},o=s.createContext(i);function t(e){const r=s.useContext(o);return s.useMemo(function(){return"function"==typeof e?e(r):{...r,...e}},[r,e])}function a(e){let r;return r=e.disableParentContext?"function"==typeof e.components?e.components(i):e.components||i:t(e.components),s.createElement(o.Provider,{value:r},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.