1"use strict";(globalThis.webpackChunkllm_d_website=globalThis.webpackChunkllm_d_website||[]).push([[6189],{49020(e,n,t){t.r(n),t.d(n,{assets:()=>a,contentTitle:()=>d,default:()=>p,frontMatter:()=>c,metadata:()=>s,toc:()=>i});const s=JSON.parse('{"id":"api-reference/epp-http-apis","title":"EPP HTTP APIs Reference","description":"This document lists the HTTP APIs the Endpoint Picker (EPP) supports for inference traffic. Depending on the API, the EPP may parse fields from the request body to do prefix-cache aware routing, and plugin decisions.","source":"@site/versioned_docs/version-0.8/api-reference/epp-http-apis.md","sourceDirName":"api-reference","slug":"/api-reference/epp-http-apis","permalink":"/docs/0.8/api-reference/epp-http-apis","draft":false,"unlisted":false,"tags":[],"version":"0.8","frontMatter":{},"sidebar":"docsSidebar","previous":{"title":"Component Config: EndpointPickerConfig","permalink":"/docs/0.8/api-reference/endpointpickerconfig"},"next":{"title":"RPC APIs","permalink":"/docs/0.8/api-reference/epp-grpc-apis"}}');var o=t(74848),l=t(28453);const c={},d="EPP HTTP APIs Reference",a={},i=[{value:"Supported HTTP APIs",id:"supported-http-apis",level:2},{value:"Request Examples",id:"request-examples",level:2},{value:"OpenAI <code>/v1/completions</code>",id:"openai-v1completions",level:3},{value:"OpenAI <code>/v1/chat/completions</code>",id:"openai-v1chatcompletions",level:3},{value:"OpenAI <code>/v1/responses</code>",id:"openai-v1responses",level:3},{value:"OpenAI <code>/v1/embeddings</code>",id:"openai-v1embeddings",level:3},{value:"Anthropic <code>/v1/messages</code>",id:"anthropic-v1messages",level:3},{value:"vLLM <code>/inference/v1/generate</code>",id:"vllm-inferencev1generate",level:3}];function r(e){const n={a:"a",code:"code",details:"details",h1:"h1",h2:"h2",h3:"h3",header:"header",hr:"hr",p:"p",pre:"pre",summary:"summary",table:"table",tbody:"tbody",td:"td",th:"th",thead:"thead",tr:"tr",...(0,l.R)(),...e.components};return(0,o.jsxs)(o.Fragment,{children:[(0,o.jsx)(n.header,{children:(0,o.jsx)(n.h1,{id:"epp-http-apis-reference",children:"EPP HTTP APIs Reference"})}),"\n",(0,o.jsxs)(n.p,{children:["This document lists the HTTP APIs the ",(0,o.jsx)(n.a,{href:"../architecture/core/router/epp",children:"Endpoint Picker (EPP)"})," supports for inference traffic. Depending on the API, the EPP may parse fields from the request body to do prefix-cache aware routing, and plugin decisions."]}),"\n",(0,o.jsx)(n.h2,{id:"supported-http-apis",children:"Supported HTTP APIs"}),"\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n",(0,o.jsxs)(n.table,{children:[(0,o.jsx)(n.thead,{children:(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.th,{children:"Endpoint"}),(0,o.jsx)(n.th,{children:"Source"}),(0,o.jsx)(n.th,{children:"Supported"})]})}),(0,o.jsxs)(n.tbody,{children:[(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"/v1/completions"})}),(0,o.jsx)(n.td,{children:"OpenAI Completions API"}),(0,o.jsx)(n.td,{children:"\u2705"})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"/v1/chat/completions"})}),(0,o.jsx)(n.td,{children:"OpenAI Chat Completions API"}),(0,o.jsx)(n.td,{children:"\u2705"})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"/v1/responses"})}),(0,o.jsx)(n.td,{children:"OpenAI Responses API"}),(0,o.jsx)(n.td,{children:"\u2705"})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"/v1/embeddings"})}),(0,o.jsx)(n.td,{children:"OpenAI Embeddings API"}),(0,o.jsx)(n.td,{children:"\u2705"})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"/v1/messages"})}),(0,o.jsx)(n.td,{children:"Anthropic Messages API"}),(0,o.jsx)(n.td,{children:"\u2705"})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"/inference/v1/generate"})}),(0,o.jsx)(n.td,{children:"vLLM Generate API"}),(0,o.jsx)(n.td,{children:"\u2705"})]})]})]}),"\n",(0,o.jsx)(n.hr,{}),"\n",(0,o.jsx)(n.h2,{id:"request-examples",children:"Request Examples"}),"\n",(0,o.jsxs)(n.p,{children:["The examples below parameterize the model as ",(0,o.jsx)(n.code,{children:"${MODEL_NAME}"})," and the proxy endpoint as ",(0,o.jsx)(n.code,{children:"${IP}"}),". Set ",(0,o.jsx)(n.code,{children:"${MODEL_NAME}"})," to ",(0,o.jsx)(n.a,{href:"https://huggingface.co/Qwen/Qwen3-VL-32B-Instruct",children:(0,o.jsx)(n.code,{children:"Qwen/Qwen3-VL-32B-Instruct"})})," from the ",(0,o.jsx)(n.a,{href:"https://github.com/llm-d/llm-d/tree/main/guides/multimodal-serving/aggregation",children:"multimodal aggregation guide"}),", and set ",(0,o.jsx)(n.code,{children:"${IP}"})," to the proxy endpoint IP retrieved per that guide's verification steps."]}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:"export MODEL_NAME=Qwen/Qwen3-VL-32B-Instruct\n"})}),"\n",(0,o.jsxs)(n.p,{children:["The ",(0,o.jsx)(n.code,{children:"/v1/embeddings"})," section overrides ",(0,o.jsx)(n.code,{children:"${MODEL_NAME}"}
1)," since chat/instruct models do not expose that route."]}),"\n",(0,o.jsxs)(n.h3,{id:"openai-v1completions",children:["OpenAI ",(0,o.jsx)(n.code,{children:"/v1/completions"})]}),"\n",(0,o.jsx)(n.p,{children:"Request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -X POST http://${IP}/v1/completions \\\n -H \'Content-Type: application/json\' \\\n -d \'{\n "model": "\'"${MODEL_NAME}"\'",\n "prompt": "Hello",\n "max_tokens": 10\n }\' | jq\n'})}),"\n",(0,o.jsx)(n.p,{children:"Response:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-json",children:'{\n "id": "cmpl-abc123",\n "object": "text_completion",\n "created": 1781036021,\n "model": "Qwen/Qwen3-VL-32B-Instruct",\n "choices": [\n {\n "index": 0,\n "text": "! I am trying to write a story, and",\n "logprobs": null,\n "finish_reason": "length",\n "stop_reason": null\n }\n ],\n "system_fingerprint": "vllm-0.21.0-tp2-5054d0df",\
1n "usage": {\n "prompt_tokens": 1,\n "total_tokens": 11,\n "completion_tokens": 10\n }\n}\n'})}),"\n",(0,o.jsxs)(n.p,{children:["Streaming request (set ",(0,o.jsx)(n.code,{children:"stream: true"}),"; the response is server-sent events, so drop ",(0,o.jsx)(n.code,{children:"jq"})," and use ",(0,o.jsx)(n.code,{children:"curl -N"})," to flush chunks as they arrive):"]}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -N -X POST http://${IP}/v1/completions \\\n -H \'Content-Type: application/json\' \\\n -d \'{\n "model": "\'"${MODEL_NAME}"\'",\n "prompt": "Hello",\n "max_tokens": 10,\n "stream": true\n }\'\n'})}),"\n",(0,o.jsxs)(n.h3,{id:"openai-v1chatcompletions",children:["OpenAI ",(0,o.jsx)(n.code,{children:"/v1/chat/completions"})]}),"\n",(0,o.jsx)(n.p,{children:"Request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -X POST http://${IP}/v1/chat/completions \\\n -H \'Content-Type: application/json\' \\\n -d \'{\n "model": "\'"${MODEL_NAME}"\'",\n "messages": [\n {\n "role": "user",\n "content": [\n {"type": "text", "text": "Describe this image."},\n {"type": "image_url", "image_url": {"url": "https://picsum.photos/640/360"}}\n ]\n }\n ],\n "max_tokens": 10\n }\' | jq\n'})}),"\n",(0,o.jsx)(n.p,{children:"Streaming request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -N -X POST http://${IP}/v1/chat/completions \\\n -H \'Content-Type: application/json\' \\\n -d \'{\n "model": "\'"${MODEL_NAME}"\'",\n "messages": [\n {\n "role": "user",\n "content": [\n {"type": "text", "text": "Describe this image."},\n {"type": "image_url", "image_url": {"url": "https://picsum.photos/640/360"}}\n ]\n }\n ],\n "max_tokens": 10,\n "stream": true\n }\'\n'})}),"\n",(0,o.jsxs)(n.details,{children:["\n",(0,o.jsx)(n.summary,{children:"Streaming response (SSE)"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{children:'data: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":"!","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":" I","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":"\'m","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":" a","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":" student","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":" of","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":" the","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":" ","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":"1","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":"0","logprobs":null,"finish_reason":"length","stop_reason":null}],"system_fingerprint":"vllm-0.21.0-tp2-5054d0df"}\n\ndata: [DONE]\n'})}),"\n"]}),"\n",(0,o.jsxs)(n.h3,{id:"openai-v1responses",children:["OpenAI ",(0,o.jsx)(n.code,{children:"/v1/responses"})]}),"\n",(0,o.jsx)(n.p,{children:"Request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -X POST http://${IP}/v1/responses \\\n -H \'Content-Type: application/json\' \\\n -d \'{\n "model": "\'"${MODEL_NAME}"\'",\n "input": "Hello",\n "max_output_tokens": 10\n }\' | jq\n'})}),"\n",(0,o.jsx)(n.p,{children:"Response:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-json",children:'{\n "id": "resp_abc127",\n "created_at": 1781036107,\n "incomplete_details": {"reason": "max_output_tokens"},\n "model": "Qwen/Qwen3-VL-32B-Instruct",\n "object": "response",\n "output": [\n {\n "id": "msg_abc128",\n "type": "message",\n "role": "assistant",\n "status": "completed",\n "content": [\n {\n "type": "output_text",\n "text": "Hello! How can I help you today?",\n "annotations": []\n }\n ]\n }\n ],\n "status": "incomplete",\n "max_output_tokens": 10,\n "usage": {\n "input_tokens": 9,\n "output_tokens": 10,\n "total_tokens": 19\n }\n}\n'})}),"\n",(0,o.jsxs)(n.h3,{id:"openai-v1embeddings",children:["OpenAI ",(0,o.jsx)(n.code,{children:"/v1/embeddings"})]}),"\n",(0,o.jsxs)(n.p,{children:["This endpoint requires an embedding model deployment (for example ",(0,o.jsx)(n.code,{children:"Qwen/Qwen3-Embedding-0.6B"}
1),"). Chat/instruct models do not expose this route."]}),"\n",(0,o.jsx)(n.p,{children:"Request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'export MODEL_NAME=Qwen/Qwen3-Embedding-0.6B\ncurl -X POST http://${IP}/v1/embeddings \\\n -H \'Content-Type: application/json\' \\\n -d \'{\n "model": "\'"${MODEL_NAME}"\'",\n "input": "Hello"\n }\' | jq\n'})}),"\n",(0,o.jsx)(n.p,{children:"Response (embedding vector truncated for readability):"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-json",children:'{\n "model": "Qwen/Qwen3-Embedding-0.6B",\n "object": "list",\n "data": [\n {\n "index": 0,\n "object": "embedding",\n "embedding": [-0.01350, -0.02152, -0.01368, -0.03032, 0.00941, "..."]\n }\n ],\n "usage": {\n "prompt_tokens": 2,\n "total_tokens": 2,\n "completion_tokens": 0\n }\n}\n'})}),"\n",(0,o.jsxs)(n.h3,{id:"anthropic-v1messages",children:["Anthropic ",(0,o.jsx)(n.code,{children:"/v1/messages"})]}),"\n",(0,o.jsx)(n.p,{children:"Request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -X POST http://${IP}/v1/messages \\\n -H \'Content-Type: application/json\' \\\n -d \'{\n "model": "\'"${MODEL_NAME}"\'",\n "messages": [\n {\n "role": "user",\n "content": [\n {"type": "text", "text": "Describe this image."},\n {"type": "image", "source": {"type": "url", "url": "https://picsum.photos/640/360"}}\n ]\n }\n ],\n "max_tokens": 10\n }\' | jq\n'})}),"\n",(0,o.jsx)(n.p,{children:"Response:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-json",children:'{\n "id": "chatcmpl-abc125",\n "type": "message",\n "role": "assistant",\n "content": [\n {\n "type": "text",\n "text": "This image is a close-up, shallow-focus photograph"\n }\n ],\n "model": "Qwen/Qwen3-VL-32B-Instruct",\n "stop_reason": "max_tokens",\n "usage": {\n "input_tokens": 234,\n "output_tokens": 10\n }\n}\n'})}),"\n",(0,o.jsx)(n.p,{children:"Streaming request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -N -X POST http://${IP}/v1/messages \\\n -H \'Content-Type: application/json\' \\\n -d \'{\n "model": "\'"${MODEL_NAME}"\'",\n "messages": [\n {\n "role": "user",\n "content": [\n {"type": "text", "text": "Describe this image."},\n {"type": "image", "source": {"type": "url", "url": "https://picsum.photos/640/360"}}\n ]\n }\n ],\n "max_tokens": 10,\n "stream": true\n }\'\n'})}),"\n",(0,o.jsxs)(n.details,{children:["\n",(0,o.jsx)(n.summary,{children:"Streaming response (SSE)"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{children:'event: message_start\ndata: {"type":"message_start","message":{"i
1d":"chatcmpl-abc126","content":[],"model":"Qwen/Qwen3-VL-32B-Instruct","stop_reason":null,"stop_sequence":null,"usage":{"input_tokens":234,"output_tokens":0}}}\n\nevent: content_block_start\ndata: {"type":"content_block_start","content_block":{"type":"text","text":""},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":"This"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" image"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" captures"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" a"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" serene"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" and"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" atmospheric"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" urban"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" landscape"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" at"},"index":0}\n\nevent: content_block_stop\ndata: {"type":"content_block_stop","index":0}\n\nevent: message_delta\ndata: {"type":"message_delta","delta":{"stop_reason":"max_tokens"},"usage":{"input_tokens":234,"output_tokens":10}}\n\nevent: message_stop\ndata: {"type":"message_stop"}\n'})}),"\n"]}),"\n",(0,o.jsxs)(n.h3,{id:"vllm-inferencev1generate",children:["vLLM ",(0,o.jsx)(n.code,{children:"/inference/v1/generate"})]}),"\n",(0,o.jsxs)(n.p,{children:["This endpoint requires the model server to be vLLM. Sampling controls must be nested inside a ",(0,o.jsx)(n.code,{children:"sampling_params"})," object rather than placed at the top level."]}),"\n",(0,o.jsx)(n.p,{children:"Request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -X POST http://${IP}/inference/v1/generate \\\n -H \'Content-Type: application/json\' \\\n -d \'{\n "model": "\'"${MODEL_NAME}"\'",\n "token_ids": [9906],\n "sampling_params": {"max_tokens": 10}\n }\' | jq\n'})}),"\n",(0,o.jsx)(n.p,{children:"Response:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-json",children:'{\n "request_id": "abc129",\n "choices": [\n {\n "index": 0,\n "logprobs": null,\n "finish_reason": "length",\n "token_ids": [17993, 1894, 7332, 198, 286, 2415, 1140, 259, 4580, 892]\n }\n ]\n}\n'})})]})}function p(e={}){const{wrapper:n}={...(0,l.R)(),...e.components};return n?(0,o.jsx)(n,{...e,children:(0,o.jsx)(r,{...e})}):r(e)}},28453(e,n,t){t.d(n,{R:()=>c,x:()=>d});var s=t(96540);const o={},l=s.createContext(o);function c(e){const n=s.useContext(l);return s.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function d(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(o):e.components||o:c(e.components),s.createElement(l.Provider,{value:n},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.