1"use strict";(self.webpackChunksemantic_router_docs=self.webpackChunksemantic_router_docs||[]).push([[25166],{6316(e,n,t){t.r(n),t.d(n,{default:()=>l,metadata:()=>i});var i=t(75774),s=t(74848),r=t(28453);const o={authorsImageUrls:[void 0]};function a(e){const n={a:"a",blockquote:"blockquote",h2:"h2",h3:"h3",hr:"hr",img:"img",li:"li",ol:"ol",p:"p",strong:"strong",ul:"ul",...(0,r.R)(),...e.components};return(0,s.jsxs)(s.Fragment,{children:[(0,s.jsx)(n.h2,{id:"introduction",children:"Introduction"}),"\n",(0,s.jsxs)(n.p,{children:["Over the past several months, AMD and the vLLM SR Team have been collaborating to bring ",(0,s.jsx)(n.strong,{children:"vLLM Semantic Router (VSR)"})," to AMD GPUs\u2014not just as a performance optimization, but as a fundamental shift in how we think about AI system architecture."]}),"\n",(0,s.jsxs)(n.p,{children:["AMD has been a long-term technology partner for the vLLM community, from accelerating the vLLM inference engine on AMD GPUs and ROCm\u2122 Software to now co-building the next layer of the AI stack: ",(0,s.jsx)(n.strong,{children:"intelligent routing and governance for Mixture-of-Models (MoM) systems"}),"."]}),"\n",(0,s.jsxs)(n.p,{children:['As AI moves from single models to multi-model architectures, the challenge is no longer "how big is your model" but ',(0,s.jsx)(n.strong,{children:"how intelligently and safely you orchestrate many models together"}),". VSR is designed to be the ",(0,s.jsx)(n.strong,{children:"intelligent control plane"})," for this new era\u2014making routing decisions based on semantic understanding, enforcing safety policies, and maintaining trust as systems scale toward AGI-level capabilities."]}),"\n",(0,s.jsx)(n.p,{children:(0,s.jsx)(n.img,{alt:"AMD \xd7 vLLM Semantic Router: Building the System Intelligence Together: Amd 0",src:t(27331).A+"",width:"1024",height:"528"})}),"\n",(0,s.jsx)(n.p,{children:"This collaboration focuses on three strategic pillars:"}),"\n",(0,s.jsxs)(n.ol,{children:["\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Signal-Based Routing"}),": Intelligent request routing using keyword matching, domain classification, semantic similarity, and fact-checking for Multi-LoRA and multi-model deployments"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Cross-Instance Intelligence"}),": Shared state and optimization across vLLM instances through centralized response storage and semantic caching"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Guardrails & Governance"}),": Enterprise-grade security from PII detection and jailbreak prevention to hallucination detection and alignment enforcement"]}),"\n"]}),"\n",(0,s.jsxs)(n.p,{children:["Together with AMD, we're building VSR to run efficiently on AMD GPUs while establishing a new standard for ",(0,s.jsx)(n.strong,{children:"trustworthy, governable AI infrastructure"}),"."]}),"\n",(0,s.jsx)(n.h2,{id:"the-shift-from-single-models-to-mixture-of-models",children:"The Shift: From Single Models to Mixture-of-Models"}),"\n",(0,s.jsx)(n.p,{children:"In a Mixture-of-Models world, an enterprise AI stack typically includes:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Router SLMs"})," (small language models) that classify, route, and enforce policy"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Multiple LLMs"})," and domain-specific models (e.g., code, finance, healthcare, legal)"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Tools, RAG pipelines"}),", vector search, and business systems"]}),"\n"]}),"\n",(0,s.jsxs)(n.p,{children:["Without a robust routing layer, this becomes an opaque and fragile mesh. The AMD \xd7 VSR collaboration aims to make routing a ",(0,s.jsx)(n.strong,{children:"first-class, GPU-accelerated infrastructure component"}),"\u2014not an ad-hoc script glued between services."]}),"\n",(0,s.jsx)(n.h2,{id:"vsr-core-capabilities",children:"VSR Core Capabilities"}),"\n",(0,s.jsx)(n.h3,{id:"1-signal-based-routing-for-multi-lora-deployments",children:"1. Signal-Based Routing for Multi-LoRA Deployments"}),"\n",(0,s.jsx)(n.p,{children:"VSR provides multiple routing strategies to match different use cases:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Keyword-based routing"}),": Simple pattern matching for fast, deterministic routing"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Domain classification"}),": Intent-aware adapter selection using trained classifiers"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Embedding-based semantic similarity"}
1),": Nuanced routing based on semantic understanding"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Fact-checking and verification routing"}),": High-stakes queries routed to specialized verification pipelines"]}),"\n"]}),"\n",(0,s.jsx)(n.h3,{id:"2-cross-instance-intelligence",children:"2. Cross-Instance Intelligence"}),"\n",(0,s.jsx)(n.p,{children:"VSR enables shared state and optimization across all vLLM instances:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Response API"}),": Centralized response storage enabling stateful multi-turn conversations"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Semantic Cache"}),": Significant token reduction through cross-instance vector similarity matching"]}),"\n"]}),"\n",(0,s.jsx)(n.h3,{id:"3-enterprise-grade-guardrails",children:"3. Enterprise-Grade Guardrails"}),"\n",(0,s.jsx)(n.p,{children:"From single-turn to multi-turn conversations, VSR provides:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"PII Detection"}),": Prevent sensitive information leakage"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Jailbreak Prevention"}),": Block malicious prompt injection attempts"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Hallucination Detection"}),": Verify response reliability for critical domains"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Super Alignment"}),": Ensuring AI systems remain aligned with human values and intentions as they scale toward AGI capabilities"]}),"\n"]}),"\n",(0,s.jsx)(n.hr,{}),"\n",(0,s.jsx)(n.h2,{id:"running-vsr-on-amd-gpus-two-deployment-paths",children:"Running VSR on AMD GPUs: Two Deployment Paths"}),"\n",(0,s.jsxs)(n.p,{children:["Our near-term objective is execution-oriented: ",(0,s.jsx)(n.strong,{children:"deliver a production-grade VSR solution that runs efficiently on AMD GPUs"}),". We're building two complementary deployment paths:"]}),"\n",(0,s.jsx)(n.p,{children:(0,s.jsx)(n.img,{alt:"AMD \xd7 vLLM Semantic Router: Building the System Intelligence Together: Amd 1",src:t(79578).A+"",width:"4800",height:"3584"})}),"\n",(0,s.jsx)(n.h3,{id:"path-1-vllm-based-inference-on-amd-gpus",children:"Path 1: vLLM-Based Inference on AMD GPUs"}),"\n",(0,s.jsx)(n.p,{children:"Using the vLLM engine on AMD GPUs, we run:"}),"\n",(0,s.jsxs)(n.p,{children:[(0,s.jsx)(n.strong,{children:"Router SLMs"})," for:"]}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"Task and intent classification"}),"\n",(0,s.jsx)(n.li,{children:"Risk scoring and safety gating"}),"\n",(0,s.jsx)(n.li,{children:"Tool and workflow selection"}),"\n"]}),"\n",(0,s.jsxs)(n.p,{children:[(0,s.jsx)(n.strong,{children:"LLMs and specialized models"})," for:"]}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"General assistance"}),"\n",(0,s.jsx)(n.li,{children:"Domain-specific tasks (finance, legal, code, healthcare)"}),"\n"]}),"\n",(0,s.jsxs)(n.p,{children:["VSR sits above as the decision fabric, consuming semantic similarity, business metadata, latency constraints, and compliance requirements to perform ",(0,s.jsx)(n.strong,{children:"dynamic routing"})," across models and endpoints."]}),"\n",(0,s.jsxs)(n.p,{children:["AMD GPUs provide the throughput and memory footprint needed to run ",(0,s.jsx)(n.strong,{children:"router SLMs + multiple LLMs"})," in the same cluster, supporting high-QPS workloads with stable latency\u2014not just one-off demos."]}),"\n",(0,s.jsx)(n.h3,{id:"path-2-lightweight-onnx-based-routing",children:"Path 2: Lightweight ONNX-Based Routing"}),"\n",(0,s.jsx)(n.p,{children:"Not all routing needs a full inference stack. For ultra-high-frequency, latency-sensitive stages at the \u201cfront door\u201d of the system, we're enabling:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsxs)(n.li,{children:["Exporting router SLMs to ",(0,s.jsx)(n.strong,{children:"ONNX"})]}),"\n",(0,s.jsx)(n.li,{children:"Running them on AMD GPUs through ONNX Runtime"}),"\n",(0,s.jsx)(n.li,{children:"Forwarding complex generative work to vLLM or other back-end LLMs"}),"\n"]}),"\n",(0,s.jsx)(n.p,{children:"This lightweight path is designed for:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"Front-of-funnel traffic classification and triage"}),"\n",(0,s.jsx)(n.li,{children:"Large-scale policy evaluation and offline experiments"}),"\n",(0,s.jsxs)(n.li,{children:["Enterprises that want to ",(0,s.jsx)(n.strong,{children:"standardize on AMD GPUs while keeping model providers flexible"})]}),"\n"]}),"\n",(0,s.jsx)(n.h2,{id:"moving-to-the-next-stage-of-semantic-router",children:"Moving to the Next Stage of Semantic Router"}),"\n",(0,s.jsxs)(n.p,{children:["When we first built vLLM Semantic Router, the goal was clear and practical: ",(0,s.jsx)(n.strong,{children:"intelligent model selection"}),"\u2014routing requests to the right model based on task type, cost constraints, and performance requirements."]}),"\n",(0,s.jsx)(n.p,{children:(0,s.jsx)(n.img,{alt:"AMD \xd7 vLLM Semantic Router: Building the System Intelligence Together: Amd 2",src:t(689).A+"",width:"3442",height:"1935"})}),"\n",(0,s.jsxs)(n.p,{children:[(0,s.jsx)(n.strong,{children:"vLLM Engine"})," delivers the foundation\u2014running large models stably and efficie
1ntly. ",(0,s.jsx)(n.strong,{children:"vLLM Semantic Router"})," provides the scheduler\u2014dispatching requests to the right capabilities."]}),"\n",(0,s.jsx)(n.p,{children:"But as AI systems move toward AGI-level capabilities, this framing feels incomplete. It's like discussing engine efficiency without addressing brakes, traffic laws, or safety systems."}),"\n",(0,s.jsx)(n.p,{children:(0,s.jsx)(n.strong,{children:"The real challenge isn't making models more powerful\u2014it's maintaining control as they become more powerful."})}),"\n",(0,s.jsx)(n.h3,{id:"from-models-director-to-intelligence-judger",children:"From Models Director to Intelligence Judger"}),"\n",(0,s.jsxs)(n.p,{children:["Working with AMD, we've come to see Semantic Router's evolution differently. Its potential lies not just in \"routing,\" but in ",(0,s.jsx)(n.strong,{children:"governance"}),"\u2014transforming from a traffic director into an ",(0,s.jsx)(n.strong,{children:"Intelligence Control Plane"})," for the AGI era."]}),"\n",(0,s.jsxs)(n.p,{children:["This shift changes how we think about the collaboration. We're not just optimizing for throughput and latency on AMD hardware. We're building a ",(0,s.jsx)(n.strong,{children:"constitutional layer"})," for AI systems\u2014one defined by responsibilities, not just features."]}),"\n",(0,s.jsx)(n.h3,{id:"three-control-lifelines-that-must-be-secured",children:"Three Control Lifelines That Must Be Secured"}),"\n",(0,s.jsx)(n.p,{children:"As we architect VSR on AMD's infrastructure, we're designing around three critical control points that determine whether AI systems remain trustworthy at scale:"}),"\n",(0,s.jsx)(n.p,{children:(0,s.jsx)(n.img,{alt:"AMD \xd7 vLLM Semantic Router: Building the System Intelligence Together: Amd 3",src:t(90472).A+"",width:"1536",height:"1024"})}),"\n",(0,s.jsx)(n.p,{children:(0,s.jsx)(n.strong,{children:"1. World Output (Actions)"})}),"\n",(0,s.jsxs)(n.p,{children:["The most dangerous capability of powerful models isn't reasoning\u2014it's ",(0,s.jsx)(n.strong,{children:"execution"}),". Every action that changes the world (tool calls, database writes, API invocations, configuration changes) must pass through an external checkpoint before execution."]}),"\n",(0,s.jsxs)(n.p,{children:["With AMD GPUs, we can run these checkpoints ",(0,s.jsx)(n.strong,{children:"inline at production scale"}),"\u2014evaluating risk, enforcing policies, and logging decisions without becoming a bottleneck."]}),"\n",(0,s.jsx)(n.p,{children:(0,s.jsx)(n.strong,{children:"2. World Input (Inputs)"})}),"\n",(0,s.jsx)(n.p,{children:"External inputs are untrusted by default. Web pages, retrieval results, uploaded files, and plugin returns can all carry prompt injection, data poisoning, or privilege escalation attempts."}),"\n",(0,s.jsxs)(n.p,{children:["VSR on AMD infrastructure provides ",(0,s.jsx)(n.strong,{children:"border inspection"})," before data reaches the model\u2014running classifiers, sanitizers, and verification checks as a first line of defense, not an afterthought."]}),"\n",(0,s.jsx)(n.p,{children:(0,s.jsx)(n.strong,{children:"3. Long-Term State (Memory/State)"})}),"\n",(0,s.jsxs)(n.p,{children:["The hardest failures to fix aren't wrong answers\u2014they're ",(0,s.jsx)(n.strong,{children:"wrong answers that get written into long-term memory, system state, or automated workflows"}),"."]}),"\n",(0,s.jsx)(n.p,{children:"Our collaboration focuses on making state management a first-class concern: who can write, what can be written, how to undo, and how to isolate contamination. AMD's GPU infrastructure enables us to run continuous verification and rollback mechanisms that keep state trustworthy over time."}),"\n",(0,s.jsx)(n.h3,{id:"the-ultimate-question",children:"The Ultimate Question"}),"\n",(0,s.jsx)(n.p,{children:"When these three lifelines are secured, Semantic Router stops being just a model selector. It becomes the answer to a fundamental question:"}),"\n",(0,s.jsx)(n.p,{children:(0,s.jsx)(n.strong,{children:"How do we transform alignment from a training-time aspiration into a runtime institution?"})}),"\n",(0,s.jsxs)(n.p,{children:["This is what the AMD \xd7 vLLM Semantic Router collaboration is really about: building not just faster routing, but ",(0,s.jsx)(n.strong,{children:"trustworthy, governable AI infrastru
1cture"})," that can scale safely toward AGI-level capabilities."]}),"\n",(0,s.jsx)(n.h2,{id:"long-term-vision-and-ongoing-work",children:"Long-Term Vision and Ongoing Work"}),"\n",(0,s.jsx)(n.p,{children:"Our collaboration with AMD extends beyond near-term deployment to building the foundation for next-generation AI infrastructure. We're working on several long-term initiatives:"}),"\n",(0,s.jsx)(n.h3,{id:"training-a-next-generation-router-model-on-amd-gpus",children:"Training a Next-Generation Router Model on AMD GPUs"}),"\n",(0,s.jsxs)(n.p,{children:["As a longer-term goal, we aim to explore training a ",(0,s.jsx)(n.strong,{children:"next-generation router model based on encoder-only"})," on AMD GPUs, optimized for semantic routing, retrieval-augmented generation (RAG), and safety classification."]}),"\n",(0,s.jsxs)(n.p,{children:["While recent encoder models (e.g., ModernBERT) show strong performance, they remain limited in context length, multilingual coverage, and alignment with emerging long-context attention techniques. This effort focuses on advancing encoder capabilities using AMD hardware, particularly for ",(0,s.jsx)(n.strong,{children:"long-context, high-throughput representation learning"}),"."]}),"\n",(0,s.jsxs)(n.p,{children:["The outcome will be an ",(0,s.jsx)(n.strong,{children:"open encoder model"})," designed to integrate with vLLM Semantic Router and modern AI pipelines, strengthening the retrieval and routing layers of AI systems while expanding hardware-diverse training and deployment options for the community and industry."]}),"\n",(0,s.jsx)(n.h3,{id:"community-public-beta-on-amd-infrastructure",children:"Community Public Beta on AMD Infrastructure"}),"\n",(0,s.jsxs)(n.p,{children:["As part of this collaboration, each major release of vLLM Semantic Router will be accompanied by a ",(0,s.jsx)(n.strong,{children:"public beta environment"})," hosted on AMD-sponsored infrastructure, available free of charge to the community."]}),"\n",(0,s.jsx)(n.p,{children:"These public betas will allow users to:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"Validate new routing, caching, and safety features"}),"\n",(0,s.jsx)(n.li,{children:"Gain hands-on experience with Semantic Router running on AMD GPUs"}),"\n",(0,s.jsx)(n.li,{children:"Provide early feedback that helps improve performance, usability, and system design"}),"\n"]}),"\n",(0,s.jsx)(n.p,{children:"By lowering the barrier to experimentation and validation, this initiative aims to strengthen the vLLM ecosystem, accelerate real-world adoption, and ensure that new Semantic Router capabilities are shaped by community input before broader production deployment."}),"\n",(0,s.jsx)(n.h3,{id:"amd-gpu-powered-cicd-and-end-to-end-testbed",children:"AMD GPU-Powered CI/CD and End-to-End Testbed"}),"\n",(0,s.jsxs)(n.p,{children:["In the long run, we aim to use AMD GPUs to underpin how ",(0,s.jsx)(n.strong,{children:"VSR as an open-source project is built, validated, and shipped"}),", ensuring VSR works consistently well with AMD GPUs as the project grows."]}),"\n",(0,s.jsxs)(n.p,{children:["We are designing a GPU-backed ",(0,s.jsx)(n.strong,{children:"CI/CD and end-to-end testbed"})," where:"]}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"Router SLMs, LLMs, domain models, retrieval, and tools run together on AMD GPU clusters"}),"\n",(0,s.jsx)(n.li,{children:"Multi-domain, multi-risk-level datasets are replayed as traffic"}),"\n",(0,s.jsxs)(n.li,{children:["Each VSR change runs through an automated evaluation pipeline, including:","\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"Routing and policy regression tests"}),"\n",(0,s.jsx)(n.li,{children:"A/B comparisons of new vs. previous strategies"}),"\n",(0,s.jsx)(n.li,{children:"Stress tests on latency, cost, and scalability"}),"\n",(0,s.jsx)(n.li,{children:"Focused suites for hallucination mitigation and compliance behavior"}),"\n"]}),"\n"]}),"\n"]}),"\n",(0,s.jsx)(n.p,{children:"The target state is clear:"}),"\n",(0,s.jsxs)(n.blockquote,{children:["\n",(0,s.jsx)(n.p,{children:(0,s.jsx)(n.strong,{children:"Every VSR release comes with a reproducible, GPU-driven evaluation report, not just a changelog."})}),"\n"]}),"\n",(0,s.jsxs)(n.p,{children:["AMD GPUs, in this model, are not only for serving models; they are the ",(0,s.jsx)(n.strong,{children:"verification engine for the routing infrastru
1cture itself"}),"."]}),"\n",(0,s.jsx)(n.h3,{id:"an-amd-backed-mixture-of-models-playground",children:"An AMD-Backed Mixture-of-Models Playground"}),"\n",(0,s.jsxs)(n.p,{children:["In parallel, we are planning an ",(0,s.jsx)(n.strong,{children:"online Mixture-of-Models playground"})," powered by AMD GPUs, open to the community and partners."]}),"\n",(0,s.jsx)(n.p,{children:"This playground will allow users to:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"Experiment with different routing strategies and model topologies under real workloads"}),"\n",(0,s.jsx)(n.li,{children:"Observe, in a visual way, how VSR decides which model to call, when to retrieve, and when to apply additional checks or fallbacks"}),"\n",(0,s.jsxs)(n.li,{children:["Compare ",(0,s.jsx)(n.strong,{children:"quality, latency, and cost trade-offs"})," across configurations"]}),"\n"]}),"\n",(0,s.jsxs)(n.p,{children:["For model vendors, tool builders, and platform providers, this becomes a ",(0,s.jsx)(n.strong,{children:"neutral, AMD GPU-backed test environment"})," to:"]}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"Integrate their components into a MoM stack"}),"\n",(0,s.jsx)(n.li,{children:"Benchmark under realistic routing and governance constraints"}),"\n",(0,s.jsx)(n.li,{children:"Showcase capabilities within a transparent, observable system"}),"\n"]}),"\n",(0,s.jsx)(n.h2,{id:"why-this-collaboration-matters",children:"Why This Collaboration Matters"}),"\n",(0,s.jsx)(n.p,{children:"Through the AMD \xd7 vLLM Semantic Router collaboration, we are aiming beyond \u201cdoes this model run on this GPU\u201d."}),"\n",(0,s.jsx)(n.p,{children:"The joint ambitions are:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsxs)(n.li,{children:["To define a ",(0,s.jsx)(n.strong,{children:"reference architecture for intelligent, GPU-accelerated routing"})," on AMD platforms, including:","\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"vLLM-based inference paths,"}),"\n",(0,s.jsx)(n.li,{children:"ONNX-based lightweight router paths,"}),"\n",(0,s.jsx)(n.li,{children:"multi-model coordination and safety enforcement."}),"\n"]}),"\n"]}),"\n",(0,s.jsxs)(n.li,{children:["To treat routing as ",(0,s.jsx)(n.strong,{children:"trusted infrastructure"}),", supported by:","\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"GPU-powered CI/CD and end-to-end evaluation,"}),"\n",(0,s.jsx)(n.li,{children:"hallucination-aware and risk-aware policies,"}),"\n",(0,s.jsx)(n.li,{children:"online learning and adaptive strategies."}),"\n"]}),"\n"]}),"\n",(0,s.jsxs)(n.li,{children:["To provide the ecosystem with a ",(0,s.jsx)(n.strong,{children:"long-lived, AMD GPU\u2013backed MoM playground"})," where ideas, models, and routing policies can be tested and evolved in the open."]}),"\n"]}),"\n",(0,s.jsxs)(n.p,{children:["In short, this is about ",(0,s.jsx)(n.strong,{children:"co-building trustworthy, evolvable multi-model AI infrastructure"}),"\u2014with AMD GPUs as a core execution and validation layer, and vLLM Semantic Router as the intelligent control plane that makes the entire system understandable, governable, and ready for real workloads."]}),"\n",(0,s.jsxs)(n.p,{children:["The technical roadmap\u2014hallucination detection, online learning, multi-model orchestration\u2014serves this larger mission. AMD's hardware provides the execution layer. VSR provides the control plane. Together, we're building the foundation for AI systems that remain aligned not through hope, but through ",(0,s.jsx)(n.strong,{children:"architecture"}),"."]}),"\n",(0,s.jsx)(n.h2,{id:"acknowledgements",children:"Acknowledgements"}),"\n",(0,s.jsx)(n.p,{children:"We would like to thank the many talented people who have contributed to this collaboration:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"AMD"}),": Andy Luo, Haichen Zhang, and the AMD AIG Teams."]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"vLLM SR"}),": Xunzhuo Liu, Huamin Chen, Chen Wang, Yue Zhu, and the vLLM Semantic Router OSS team."]}),"\n"]}),"\n",(0,s.jsx)(n.p,{children:"We're excited to keep refining and expanding our optimizations to unlock even greater capabilities in the weeks and months ahead!"}),"\n",(0,s.jsx)(n.h2,{id:"join-us",children:"Join Us"}),"\n",(0,s.jsxs)(n.p,{children:[(0,s.jsx)(n.strong,{children:"Looking for Collaborations!"})," Calling all passionate community developers and researchers: join us in training the next-generation router model on AMD GPUs and building the future of trustworthy AI infrastru
1cture."]}),"\n",(0,s.jsx)(n.p,{children:"Interested? Reach out to us:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsxs)(n.li,{children:["Haichen Zhang: ",(0,s.jsx)(n.a,{href:"mailto:[email protected]",children:"[email protected]"})]}),"\n",(0,s.jsxs)(n.li,{children:["Xunzhuo Liu: ",(0,s.jsx)(n.a,{href:"mailto:[email protected]",children:"[email protected]"})]}),"\n"]}),"\n",(0,s.jsxs)(n.p,{children:[(0,s.jsx)(n.strong,{children:"Resources"}),":"]}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"https://www.amd.com/en/products/software/rocm.html",children:"AMD ROCm\u2122 Software"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"https://github.com/vllm-project/semantic-router",children:"vLLM Semantic Router GitHub Repo"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"https://vllm-sr.ai",children:"vLLM Semantic Router Documentation"})}),"\n"]}),"\n",(0,s.jsxs)(n.p,{children:[(0,s.jsx)(n.strong,{children:"Join the discussion"}),": Share your use cases and feedback in #semantic-router channel on ",(0,s.jsx)(n.a,{href:"https://vllm-dev.slack.com/archives/C09CTGF8KCN",children:"vLLM Slack"})]})]})}function l(e={}){const{wrapper:n}={...(0,r.R)(),...e.components};return n?(0,s.jsx)(n,{...e,children:(0,s.jsx)(a,{...e})}):a(e)}t.d(n,["assets",0,o,"contentTitle",0,void 0,"frontMatter",0,{slug:"vllm-sr-amd",title:"AMD \xd7 vLLM Semantic Router: Building the System Intelligence Together",description:"How AMD and vLLM Semantic Router build GPU-accelerated Mixture-of-Models routing with signals, semantic caching, response storage, PII, jailbreak, and hallucination guardrails.",authors:[{name:"The AMD and vLLM Semantic Router Team",url:"https://github.com/vllm-project/semantic-router"}],tags:["hardware","ecosystem"],image:"/img/blog/vllm/semantic-router/amd-0.png",source_url:"https://github.com/vllm-project/vllm-project.github.io/blob/main/_posts/2025-12-16-vllm-sr-amd.md"},"toc",0,[{value:"Introduction",id:"introduction",level:2},{value:"The Shift: From Single Models to Mixture-of-Models",id:"the-shift-from-single-models-to-mixture-of-models",level:2},{value:"VSR Core Capabilities",id:"vsr-core-capabilities",level:2},{value:"1. Signal-Based Routing for Multi-LoRA Deployments",id:"1-signal-based-routing-for-multi-lora-deployments",level:3},{value:"2. Cross-Instance Intelligence",id:"2-cross-instance-intelligence",level:3},{value:"3. Enterprise-Grade Guardrails",id:"3-enterprise-grade-guardrails",level:3},{value:"Running VSR on AMD GPUs: Two Deployment Paths",id:"running-vsr-on-amd-gpus-two-deployment-paths",level:2},{value:"Path 1: vLLM-Based Inference on AMD GPUs",id:"path-1-vllm-based-inference-on-amd-gpus",level:3},{value:"Path 2: Lightweight ONNX-Based Routing",id:"path-2-lightweight-onnx-based-routing",level:3},{value:"Moving to the Next Stage of Semantic Router",id:"moving-to-the-next-stage-of-semantic-router",level:2},{value:"From Models Director to Intelligence Judger",id:"from-models-director-to-intelligence-judger",level:3},{value:"Three Control Lifelines That Must Be Secured",id:"three-control-lifelines-that-must-be-secured",level:3},{value:"The Ultimate Question",id:"the-ultimate-question",level:3},{value:"Long-Term Vision and Ongoing Work",id:"long-term-vision-and-ongoing-work",level:2},{value:"Training a Next-Generation Router Model on AMD GPUs",id:"training-a-next-generation-router-model-on-amd-gpus",level:3},{value:"Community Public Beta on AMD Infrastructure",id:"community-public-beta-on-amd-infrastructure",level:3},{value:"AMD GPU-Powered CI/CD and End-to-End Testbed",id:"amd-gpu-powered-cicd-and-end-to-end-testbed",level:3},{value:"An AMD-Backed Mixture-of-Models Playground",id:"an-amd-backed-mixture-of-models-playground",level:3},{value:"Why This Collaboration Matters",id:"why-this-collaboration-matters",level:2},{value:"Acknowledgements",id:"acknowledgements",level:2},{value:"Join Us",id:"join-us",level:2}]])},27331(e,n,t){const i=t.p+"assets/images/amd-0-a7a699a8bab0028d51464c9f8bad4eec.png";t.d(n,["A",0,i])},79578(e,n,t){const i=t.p+"assets/images/amd-1-f4c9849e4739e56d29f2ffb98c1da217.png";t.d(n,["A",0,i])},689(e,n,t){const i=t.p+"assets/images/amd-2-d3fa14a40d77f1679f18de20998d65b2.png";t.d(n,["A",0,i])},90472(e,n,t){const i=t.p+"assets/images/amd-3-a5a7a6ebb220377eb0248324c5814ce6.png";t.d(n,["A",0,i])},28453(e,n,t){t.d(n,{R:()=>o,x:()=>a});var i=t(96540);const s={},r=i.createContext(s);function o(e){const n=i.useContext(r);return i.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function a(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(s):e.components||s:o(e.components),i.createElement(r.Provider,{value:n},e.children)}},75774(e){e.exports=JSON.parse('{"permalink":"/blog/vllm-sr-amd","editUrl":"https://github.com/vllm-project/semantic-router/tree/main/website/blog/blog/2025-12-16-vllm-sr-amd.md","source":"@site/blog/2025-12-16-vllm-sr-amd.md","title":"AMD \xd7 vLLM Semantic Router: Building the System Intelligence Together","description":"How AMD and vLLM Semantic Router build GPU-accelerated Mixture-of-Models routing with signals, semantic caching, response storage, PII, jailbreak, and hallucination guardrails.","date":"2025-12-16T00:00:00.000Z","tags":[{"inline":true,"label":"hardware","permalink":"/blog/tags/hardware"},{"inline":true,"label":"ecosystem","permalink":"/blog/tags/ecosystem"}],"readingTime":10.73,"hasTruncateMarker":false,"authors":[{"name":"The AMD and vLLM Semantic Router Team","url":"https://github.com/vllm-project/semantic-router","socials":{},"key":null,"page":null}],"frontMatter":{"slug":"vllm-sr-amd","title":"AMD \xd7 vLLM Semantic Router: Building the System Intelligence Together","description":"How AMD and vLLM Semantic Router build GPU-accelerated Mixture-of-Models routing with signals, semantic caching, response storage, PII, jailbreak, and hallucination guardrails.","authors":[{"name":"The AMD and vLLM Semantic Router Team","url":"https://github.com/vllm-project/semantic-router"}],"tags":["hardware","ecosystem"],"image":"/img/blog/vllm/semantic-router/amd-0.png","source_url":"https://github.com/vllm-project/vllm-project.github.io/blob/main/_posts/2025-12-16-vllm-sr-amd.md"},"unlisted":false,"prevItem":{"title":"vLLM Semantic Router v0.1 Iris: The First Major Release","permalink":"/blog/vllm-sr-iris"},"nextItem":{"title":"Token-Level Truth: Real-Time Hallucination Detection for Production LLMs","permalink":"/blog/halugate"}}')}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.