1"use strict";(self.webpackChunksemantic_router_docs=self.webpackChunksemantic_router_docs||[]).push([[12231],{13467(e,t,n){n.r(t),n.d(t,{assets:()=>o,contentTitle:()=>l,default:()=>u,frontMatter:()=>a,metadata:()=>i,toc:()=>c});const i=JSON.parse('{"id":"installation/k8s/aibrix","title":"Install with vLLM AIBrix","description":"This guide provides step-by-step instructions for integrating the vLLM AIBrix.","source":"@site/versioned_docs/version-v0.3/installation/k8s/aibrix.md","sourceDirName":"installation/k8s","slug":"/installation/k8s/aibrix","permalink":"/docs/v0.3/installation/k8s/aibrix","draft":false,"unlisted":false,"editUrl":"https://github.com/vllm-project/semantic-router/edit/main/website/docs/installation/k8s/aibrix.md","tags":[],"version":"v0.3","frontMatter":{},"sidebar":"tutorialSidebar","previous":{"title":"Install with vLLM Production Stack","permalink":"/docs/v0.3/installation/k8s/production-stack"},"next":{"title":"Install with LLM-D","permalink":"/docs/v0.3/installation/k8s/llm-d"}}');var s=n(74848),r=n(28453);const a={},l="Install with vLLM AIBrix",o={},c=[{value:"About vLLM AIBrix",id:"about-vllm-aibrix",level:2},{value:"Key Features",id:"key-features",level:3},{value:"Integration Benefits",id:"integration-benefits",level:3},{value:"Prerequisites",id:"prerequisites",level:2},{value:"Step 1: Create Kind Cluster (Optional)",id:"step-1-create-kind-cluster-optional",level:2},{value:"Step 2: Deploy vLLM Semantic Router",id:"step-2-deploy-vllm-semantic-router",level:2},{value:"Step 3: Install vLLM AIBrix",id:"step-3-install-vllm-aibrix",level:2},{value:"Step 4: Deploy Demo LLM",id:"step-4-deploy-demo-llm",level:2},{value:"Step 5: Create Gateway API Resources",id:"step-5-create-gateway-api-resources",level:2},{value:"Testing the Deployment",id:"testing-the-deployment",level:2},{value:"Method 1: Port Forwarding (Recommended for Local Testing)",id:"method-1-port-forwarding-recommended-for-local-testing",level:3},{value:"Send Test Requests",id:"send-test-requests",level:3},{value:"Cleanup",id:"cleanup",level:2},{value:"Next Steps",id:"next-steps",level:2}];function d(e){const t={a:"a",code:"code",h1:"h1",h2:"h2",h3:"h3",header:"header",li:"li",ol:"ol",p:"p",pre:"pre",strong:"strong",ul:"ul",...(0,r.R)(),...e.components};return(0,s.jsxs)(s.Fragment,{children:[(0,s.jsx)(t.header,{children:(0,s.jsx)(t.h1,{id:"install-with-vllm-aibrix",children:"Install with vLLM AIBrix"})}),"\n",(0,s.jsx)(t.p,{children:"This guide provides step-by-step instructions for integrating the vLLM AIBrix."}),"\n",(0,s.jsx)(t.h2,{id:"about-vllm-aibrix",children:"About vLLM AIBrix"}),"\n",(0,s.jsxs)(t.p,{children:[(0,s.jsx)(t.a,{href:"https://github.com/vllm-project/aibrix",children:"vLLM AIBrix"})," is an open-source initiative designed to provide essential building blocks to construct scalable GenAI inference infrastructure. AIBrix delivers a cloud-native solution optimized for deploying, managing, and scaling large language model (LLM) inference, tailored specifically to enterprise needs."]}),"\n",(0,s.jsx)(t.h3,{id:"key-features",children:"Key Features"}),"\n",(0,s.jsxs)(t.ul,{children:["\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.strong,{children:"High-Density LoRA Management"}),": Streamlined support for lightweight, low-rank adaptations of models"]}),"\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.strong,{children:"LLM Gateway and Routing"}),": Efficiently manage and direct traffic across multiple models and replicas"]}),"\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.strong,{children:"LLM App-Tailored Autoscaler"}),": Dynamically scale inference resources based on real-time demand"]}),"\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.strong,{children:"Unified AI Runtime"}),": A versatile sidecar enabling metric standardization, model downloading, and management"]}),"\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.strong,{children:"Distributed Inference"}),": Scalable architecture to handle large workloads across multiple nodes"]}),"\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.strong,{children:"Distributed KV Cache"}),": Enables high-capacity, cross-engine KV reuse"]}),"\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.strong,{children:"Cost-efficient Heterogeneous Serving"}),": Enables mixed GPU inference to reduce costs with SLO guarantees"]}),"\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.strong,{children:"GPU Hardware Failure Detection"}),": Proactive detection of GPU hardware issues"]}),"\n"]}),"\n",(0,s.jsx)(t.h3,{id:"integration-benefits",children:"Integration Benefits"}),"\n",(0,s.jsx)(t.p,{children:"Integrating vLLM Semantic Router with AIBrix provides several advantages:"}),"\n",(0,s.jsxs)(t.ol,{children:["\n",(0,s.jsxs)(t.li,{children:["\n",(0,s.jsxs)(t.p,{children:[(0,s.jsx)(t.strong,{children:"Intelligent Request Routing"}),": Semantic Router analyzes incoming requests and routes them to the most appropriate model based on content understanding, while AIBrix's gateway efficie
1ntly manages traffic distribution across model replicas"]}),"\n"]}),"\n",(0,s.jsxs)(t.li,{children:["\n",(0,s.jsxs)(t.p,{children:[(0,s.jsx)(t.strong,{children:"Enhanced Scalability"}),": AIBrix's autoscaler works seamlessly with Semantic Router to dynamically adjust resources based on routing patterns and real-time demand"]}),"\n"]}),"\n",(0,s.jsxs)(t.li,{children:["\n",(0,s.jsxs)(t.p,{children:[(0,s.jsx)(t.strong,{children:"Cost Optimization"}),": By combining Semantic Router's intelligent routing with AIBrix's heterogeneous serving capabilities, you can optimize GPU utilization and reduce infrastructure costs while maintaining SLO guarantees"]}),"\n"]}),"\n",(0,s.jsxs)(t.li,{children:["\n",(0,s.jsxs)(t.p,{children:[(0,s.jsx)(t.strong,{children:"Production-Ready Infrastructure"}),": AIBrix provides enterprise-grade features like distributed KV cache, GPU failure detection, and unified runtime management, making it easier to deploy Semantic Router in production environments"]}),"\n"]}),"\n",(0,s.jsxs)(t.li,{children:["\n",(0,s.jsxs)(t.p,{children:[(0,s.jsx)(t.strong,{children:"Simplified Operations"}),": The integration leverages Kubernetes-native patterns and Gateway API resources, providing a familiar operational model for DevOps teams"]}),"\n"]}),"\n"]}),"\n",(0,s.jsx)(t.h2,{id:"prerequisites",children:"Prerequisites"}),"\n",(0,s.jsx)(t.p,{children:"Before starting, ensure you have the following tools installed:"}),"\n",(0,s.jsxs)(t.ul,{children:["\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.a,{href:"https://kind.sigs.k8s.io/docs/user/quick-start/#installation",children:"kind"})," - Kubernetes in Docker (Optional)"]}),"\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.a,{href:"https://kubernetes.io/docs/tasks/tools/",children:"kubectl"})," - Kubernetes CLI"]}),"\n",(0,s.jsxs)(t.li,{children:[(0,s.jsx)(t.a,{href:"https://helm.sh/docs/intro/install/",children:"Helm"})," - Package manager for Kubernetes"]}),"\n"]}),"\n",(0,s.jsx)(t.h2,{id:"step-1-create-kind-cluster-optional",children:"Step 1: Create Kind Cluster (Optional)"}),"\n",(0,s.jsx)(t.p,{children:"Create a local Kubernetes cluster optimized for the semantic router workload:"}),"\n",(0,s.jsx)(t.pre,{children:(0,s.jsx)(t.code,{className:"language-bash",children:"kind create cluster --name semantic-router-cluster\n\n# Verify cluster is ready\nkubectl wait --for=condition=Ready nodes --all --timeout=300s\n"})}),"\n",(0,s.jsx)(t.h2,{id:"step-2-deploy-vllm-semantic-router",children:"Step 2: Deploy vLLM Semantic Router"}),"\n",(0,s.jsx)(t.p,{children:"Deploy the semantic router service with all required components using Helm:"}),"\n",(0,s.jsx)(t.pre,{children:(0,s.jsx)(t.code,{className:"language-bash",children:"# Install with custom values from GHCR OCI registry\n# (Optional) If you use a registry mirror/proxy, append: --set global.imageRegistry=<your-registry>\nhelm install semantic-router oci://ghcr.io/vllm-project/charts/semantic-router \\\n --version v0.0.0-latest \\\n --namespace vllm-semantic-router-system \\\n --create-namespace \\\n -f https://raw.githubusercontent.com/vllm-project/semantic-router/refs/heads/main/deploy/kubernetes/aibrix/semantic-router-values/values.yaml\n\n# Wait for deployment to be ready (this may take several minutes for model downloads)\nkubectl wait --for=condition=Available deployment/semantic-router -n vllm-semantic-router-system --timeout=600s\n\n# Verify deployment status\nkubectl get pods -n vllm-semantic-router-system\n"})}),"\n",(0,s.jsxs)(t.p,{children:[(0,s.jsx)(t.strong,{children:"Note"}),": The values file contains the configuration for the semantic router, including model settings, categories, and routing rules. You can download and customize it from ",(0,s.jsx)(t.a,{href:"https://raw.githubusercontent.com/vllm-project/semantic-router/refs/heads/main/deploy/kubernetes/aibrix/semantic-router-values/values.yaml",children:"values.yaml"}),"."]}),"\n",(0,s.jsx)(t.h2,{id:"step-3-install-vllm-aibrix",children:"Step 3: Install vLLM AIBrix"}),"\n",(0,s.jsx)(t.p,{children:"Install the core vLLM AIBrix components:"}),"\n",(0,s.jsx)(t.pre,{children:(0,s.jsx)(t.code,{className:"language-bash",children:"# Install vLLM AIBrix\nkubectl create -f https://github.com/vllm-project/aibrix/releases/download/v0.4.1/aibrix-dependency-v0.4.1.yaml\n\nkubectl create -f https://github.com/vllm-project/aibrix/releases/download/v0.4.1/aibrix-core-v0.4.1.yaml\n\n# wait for deployment to be ready\nkubectl wait --timeout=2m -n aibrix-system deployment/aibrix-gateway-plugins --for=condition=Available\nkubectl wait --timeout=2m -n aibrix-system deployment/aibrix-metadata-service --for=condition=Available\nkubectl wait --timeout=2m -n aibrix-system deployment/aibrix-controller-manager --for=condition=Available\n"})}),"\n",(0,s.jsx)(t.h2,{id:"step-4-deploy-demo-llm",children:"Step 4: Deploy Demo LLM"}),"\n",(0,s.jsx)(t.p,{children:"Create a demo LLM to serve as the backend for the semantic router:"}),"\n",(0,s.jsx)(t.pre,{children:(0,s.jsx)(t.code,{className:"language-bash",children:"# Deploy demo LLM\nkubectl apply -f https://raw.githubusercontent.com/vllm-project/semantic-router/refs/heads/main/deploy/kubernetes/aibrix/aigw-resources/base-model.y
1aml\n\nkubectl wait --timeout=2m -n default deployment/vllm-llama3-8b-instruct --for=condition=Available\n"})}),"\n",(0,s.jsx)(t.h2,{id:"step-5-create-gateway-api-resources",children:"Step 5: Create Gateway API Resources"}),"\n",(0,s.jsx)(t.p,{children:"Create the necessary Gateway API resources for the envoy gateway:"}),"\n",(0,s.jsx)(t.pre,{children:(0,s.jsx)(t.code,{className:"language-bash",children:"kubectl apply -f https://raw.githubusercontent.com/vllm-project/semantic-router/refs/heads/main/deploy/kubernetes/aibrix/aigw-resources/gwapi-resources.yaml\n"})}),"\n",(0,s.jsx)(t.h2,{id:"testing-the-deployment",children:"Testing the Deployment"}),"\n",(0,s.jsx)(t.h3,{id:"method-1-port-forwarding-recommended-for-local-testing",children:"Method 1: Port Forwarding (Recommended for Local Testing)"}),"\n",(0,s.jsx)(t.p,{children:"Set up port forwarding to access the gateway locally:"}),"\n",(0,s.jsx)(t.pre,{children:(0,s.jsx)(t.code,{className:"language-bash",children:"# Get the Envoy service name\nexport ENVOY_SERVICE=$(kubectl get svc -n envoy-gateway-system \\\n --selector=gateway.envoyproxy.io/owning-gateway-namespace=aibrix-system,gateway.envoyproxy.io/owning-gateway-name=aibrix-eg \\\n -o jsonpath='{.items[0].metadata.name}')\n\nkubectl port-forward -n envoy-gateway-system svc/$ENVOY_SERVICE 8080:80\n"})}),"\n",(0,s.jsx)(t.h3,{id:"send-test-requests",children:"Send Test Requests"}),"\n",(0,s.jsx)(t.p,{children:"Once the gateway is accessible, test the inference endpoint:"}),"\n",(0,s.jsx)(t.pre,{children:(0,s.jsx)(t.code,{className:"language-bash",children:'# Test math domain chat completions endpoint\ncurl -i -X POST http://localhost:8080/v1/chat/completions \\\n -H "Content-Type: application/json" \\\n -d \'{\n "model": "MoM",\n "messages": [\n {"role": "user", "content": "What is the derivative of f(x) = x^3?"}\n ]\n }\'\n'})}),"\n",(0,s.jsx)(t.p,{children:"You will see the response from the demo LLM, and additional headers injected by the semantic router."}),"\n",(0,s.jsx)(t.pre,{children:(0,s.jsx)(t.code,{className:"language-bash",children:'HTTP/1.1 200 OK\nserver: fasthttp\ndate: Thu, 06 Nov 2025 06:38:08 GMT\ncontent-type: application/json\nx-inference-pod: vllm-llama3-8b-instruct-984659dbb-gp5l9\nx-went-into-req-headers: true\nrequest-id: b46b6f7b-5645-470f-9868-0dd8b99a7163\nx-vsr-selected-category: math\nx-vsr-selected-reasoning: on\nx-vsr-selected-model: vllm-llama3-8b-instruct\nx-vsr-injected-system-prompt: true\ntransfer-encoding: chunked\n\n{"id":"chatcmpl-f390a0c6-b38f-4a73-b019-9374a3c5d69b","created":1762411088,"model":"vllm-llama3-8b-instruct","usage":{"prompt_tokens":42,"completion_tokens":48,"total_tokens":90},"object":"chat.completion","do_remote_decode":false,"do_remote_prefill":false,"remote_block_ids":null,"remote_engine_id":"","remote_host":"","remote_port":0,"choices":[{"index":0,"finish_reason":"stop","message":{"role":"assistant","content":"I am your AI assistant, how can I help you today? To be or not to be that is the question. Alas, poor Yorick! I knew him, Horatio: A fellow of infinite jest Testing, testing 1,2,3"}}]}\n'})}),"\n",(0,s.jsx)(t.h2,{id:"cleanup",children:"Cleanup"}),"\n",(0,s.jsx)(t.p,{children:"To remove the entire deployment:"}),"\n",(0,s.jsx)(t.pre,{children:(0,s.jsx)(t.code,{className:"language-bash",children:"# Remove Gateway API resources and Demo LLM\nkubectl delete -f https://raw.githubusercontent.com/vllm-project/semantic-router/refs/heads/main/deploy/kubernetes/aibrix/aigw-resources/gwapi-resources.yaml\nkubectl delete -f https://raw.githubusercontent.com/vllm-project/semantic-router/refs/heads/main/deploy/kubernetes/aibrix/aigw-resources/base-model.yaml\n\n# Remove semantic router\nhelm uninstall semantic-router -n vllm-semantic-router-system\n\n# Delete kind cluster (optional)\nkind delete cluster --name semantic-router-cluster\n"})}),"\n",(0,s.jsx)(t.h2,{id:"next-steps",children:"Next Steps"}),"\n",(0,s.jsxs)(t.ul,{children:["\n",(0,s.jsx)(t.li,{children:"Set up monitoring and observability"}),"\n",(0,s.jsx)(t.li,{children:"Implement authentication and authorization"}),"\n",(0,s.jsx)(t.li,{children:"Scale the semantic router deployment for production workloads"}),"\n"]})]})}function u(e={}){const{wrapper:t}={...(0,r.R)(),...e.components};return t?(0,s.jsx)(t,{...e,children:(0,s.jsx)(d,{...e})}):d(e)}},28453(e,t,n){n.d(t,{R:()=>a,x:()=>l});var i=n(96540);const s={},r=i.createContext(s);function a(e){const t=i.useContext(r);return i.useMemo(function(){return"function"==typeof e?e(t):{...t,...e}},[t,e])}function l(e){let t;return t=e.disableParentContext?"function"==typeof e.components?e.components(s):e.components||s:a(e.components),i.createElement(r.Provider,{value:t},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.