PageSourceSearch

https://llm-d.ai/assets/js/221226e8.a7d5ff96.js

js llm-d.ai collected 2026-10-01 10:40:00 UTC 18,538 bytes, 1 lines download raw bytes

1"use strict";(globalThis.webpackChunkllm_d_website=globalThis.webpackChunkllm_d_website||[]).push([[6189],{49020(e,n,t){t.r(n),t.d(n,{assets:()=>a,contentTitle:()=>d,default:()=>p,frontMatter:()=>c,metadata:()=>s,toc:()=>i});const s=JSON.parse('{"id":"api-reference/epp-http-apis","title":"EPP HTTP APIs Reference","description":"This document lists the HTTP APIs the Endpoint Picker (EPP) supports for inference traffic. Depending on the API, the EPP may parse fields from the request body to do prefix-cache aware routing, and plugin decisions.","source":"@site/versioned_docs/version-0.8/api-reference/epp-http-apis.md","sourceDirName":"api-reference","slug":"/api-reference/epp-http-apis","permalink":"/docs/0.8/api-reference/epp-http-apis","draft":false,"unlisted":false,"tags":[],"version":"0.8","frontMatter":{},"sidebar":"docsSidebar","previous":{"title":"Component Config: EndpointPickerConfig","permalink":"/docs/0.8/api-reference/endpointpickerconfig"},"next":{"title":"RPC APIs","permalink":"/docs/0.8/api-reference/epp-grpc-apis"}}');var o=t(74848),l=t(28453);const c={},d="EPP HTTP APIs Reference",a={},i=[{value:"Supported HTTP APIs",id:"supported-http-apis",level:2},{value:"Request Examples",id:"request-examples",level:2},{value:"OpenAI <code>/v1/completions</code>",id:"openai-v1completions",level:3},{value:"OpenAI <code>/v1/chat/completions</code>",id:"openai-v1chatcompletions",level:3},{value:"OpenAI <code>/v1/responses</code>",id:"openai-v1responses",level:3},{value:"OpenAI <code>/v1/embeddings</code>",id:"openai-v1embeddings",level:3},{value:"Anthropic <code>/v1/messages</code>",id:"anthropic-v1messages",level:3},{value:"vLLM <code>/inference/v1/generate</code>",id:"vllm-inferencev1generate",level:3}];function r(e){const n={a:"a",code:"code",details:"details",h1:"h1",h2:"h2",h3:"h3",header:"header",hr:"hr",p:"p",pre:"pre",summary:"summary",table:"table",tbody:"tbody",td:"td",th:"th",thead:"thead",tr:"tr",...(0,l.R)(),...e.components};return(0,o.jsxs)(o.Fragment,{children:[(0,o.jsx)(n.header,{children:(0,o.jsx)(n.h1,{id:"epp-http-apis-reference",children:"EPP HTTP APIs Reference"})}),"\n",(0,o.jsxs)(n.p,{children:["This document lists the HTTP APIs the ",(0,o.jsx)(n.a,{href:"../architecture/core/router/epp",children:"Endpoint Picker (EPP)"})," supports for inference traffic. Depending on the API, the EPP may parse fields from the request body to do prefix-cache aware routing, and plugin decisions."]}),"\n",(0,o.jsx)(n.h2,{id:"supported-http-apis",children:"Supported HTTP APIs"}),"\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n",(0,o.jsxs)(n.table,{children:[(0,o.jsx)(n.thead,{children:(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.th,{children:"Endpoint"}),(0,o.jsx)(n.th,{children:"Source"}),(0,o.jsx)(n.th,{children:"Supported"})]})}),(0,o.jsxs)(n.tbody,{children:[(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"/v1/completions"})}),(0,o.jsx)(n.td,{children:"OpenAI Completions API"}),(0,o.jsx)(n.td,{children:"\u2705"})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"/v1/chat/completions"})}),(0,o.jsx)(n.td,{children:"OpenAI Chat Completions API"}),(0,o.jsx)(n.td,{children:"\u2705"})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"/v1/responses"})}),(0,o.jsx)(n.td,{children:"OpenAI Responses API"}),(0,o.jsx)(n.td,{children:"\u2705"})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"/v1/embeddings"})}),(0,o.jsx)(n.td,{children:"OpenAI Embeddings API"}),(0,o.jsx)(n.td,{children:"\u2705"})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"/v1/messages"})}),(0,o.jsx)(n.td,{children:"Anthropic Messages API"}),(0,o.jsx)(n.td,{children:"\u2705"})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"/inference/v1/generate"})}),(0,o.jsx)(n.td,{children:"vLLM Generate API"}),(0,o.jsx)(n.td,{children:"\u2705"})]})]})]}),"\n",(0,o.jsx)(n.hr,{}),"\n",(0,o.jsx)(n.h2,{id:"request-examples",children:"Request Examples"}),"\n",(0,o.jsxs)(n.p,{children:["The examples below parameterize the model as ",(0,o.jsx)(n.code,{children:"${MODEL_NAME}"})," and the proxy endpoint as ",(0,o.jsx)(n.code,{children:"${IP}"}),". Set ",(0,o.jsx)(n.code,{children:"${MODEL_NAME}"})," to ",(0,o.jsx)(n.a,{href:"https://huggingface.co/Qwen/Qwen3-VL-32B-Instruct",children:(0,o.jsx)(n.code,{children:"Qwen/Qwen3-VL-32B-Instruct"})})," from the ",(0,o.jsx)(n.a,{href:"https://github.com/llm-d/llm-d/tree/main/guides/multimodal-serving/aggregation",children:"multimodal aggregation guide"}),", and set ",(0,o.jsx)(n.code,{children:"${IP}"})," to the proxy endpoint IP retrieved per that guide's verification steps."]}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:"export MODEL_NAME=Qwen/Qwen3-VL-32B-Instruct\n"})}),"\n",(0,o.jsxs)(n.p,{children:["The ",(0,o.jsx)(n.code,{children:"/v1/embeddings"})," section overrides ",(0,o.jsx)(n.code,{children:"${MODEL_NAME}"}
1)," since chat/instruct models do not expose that route."]}),"\n",(0,o.jsxs)(n.h3,{id:"openai-v1completions",children:["OpenAI ",(0,o.jsx)(n.code,{children:"/v1/completions"})]}),"\n",(0,o.jsx)(n.p,{children:"Request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -X POST http://${IP}/v1/completions \\\n    -H \'Content-Type: application/json\' \\\n    -d \'{\n        "model": "\'"${MODEL_NAME}"\'",\n        "prompt": "Hello",\n        "max_tokens": 10\n    }\' | jq\n'})}),"\n",(0,o.jsx)(n.p,{children:"Response:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-json",children:'{\n  "id": "cmpl-abc123",\n  "object": "text_completion",\n  "created": 1781036021,\n  "model": "Qwen/Qwen3-VL-32B-Instruct",\n  "choices": [\n    {\n      "index": 0,\n      "text": "! I am trying to write a story, and",\n      "logprobs": null,\n      "finish_reason": "length",\n      "stop_reason": null\n    }\n  ],\n  "system_fingerprint": "vllm-0.21.0-tp2-5054d0df",\
1n  "usage": {\n    "prompt_tokens": 1,\n    "total_tokens": 11,\n    "completion_tokens": 10\n  }\n}\n'})}),"\n",(0,o.jsxs)(n.p,{children:["Streaming request (set ",(0,o.jsx)(n.code,{children:"stream: true"}),"; the response is server-sent events, so drop ",(0,o.jsx)(n.code,{children:"jq"})," and use ",(0,o.jsx)(n.code,{children:"curl -N"})," to flush chunks as they arrive):"]}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -N -X POST http://${IP}/v1/completions \\\n    -H \'Content-Type: application/json\' \\\n    -d \'{\n        "model": "\'"${MODEL_NAME}"\'",\n        "prompt": "Hello",\n        "max_tokens": 10,\n        "stream": true\n    }\'\n'})}),"\n",(0,o.jsxs)(n.h3,{id:"openai-v1chatcompletions",children:["OpenAI ",(0,o.jsx)(n.code,{children:"/v1/chat/completions"})]}),"\n",(0,o.jsx)(n.p,{children:"Request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -X POST http://${IP}/v1/chat/completions \\\n    -H \'Content-Type: application/json\' \\\n    -d \'{\n        "model": "\'"${MODEL_NAME}"\'",\n        "messages": [\n            {\n                "role": "user",\n                "content": [\n                    {"type": "text", "text": "Describe this image."},\n                    {"type": "image_url", "image_url": {"url": "https://picsum.photos/640/360"}}\n                ]\n            }\n        ],\n        "max_tokens": 10\n    }\' | jq\n'})}),"\n",(0,o.jsx)(n.p,{children:"Streaming request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -N -X POST http://${IP}/v1/chat/completions \\\n    -H \'Content-Type: application/json\' \\\n    -d \'{\n        "model": "\'"${MODEL_NAME}"\'",\n        "messages": [\n            {\n                "role": "user",\n                "content": [\n                    {"type": "text", "text": "Describe this image."},\n                    {"type": "image_url", "image_url": {"url": "https://picsum.photos/640/360"}}\n                ]\n            }\n        ],\n        "max_tokens": 10,\n        "stream": true\n    }\'\n'})}),"\n",(0,o.jsxs)(n.details,{children:["\n",(0,o.jsx)(n.summary,{children:"Streaming response (SSE)"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{children:'data: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":"!","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":" I","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":"\'m","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":" a","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":" student","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":" of","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":" the","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":" ","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":"1","logprobs":null,"finish_reason":null,"stop_reason":null}]}\n\ndata: {"id":"cmpl-abc124","object":"text_completion","created":1781036045,"model":"Qwen/Qwen3-VL-32B-Instruct","choices":[{"index":0,"text":"0","logprobs":null,"finish_reason":"length","stop_reason":null}],"system_fingerprint":"vllm-0.21.0-tp2-5054d0df"}\n\ndata: [DONE]\n'})}),"\n"]}),"\n",(0,o.jsxs)(n.h3,{id:"openai-v1responses",children:["OpenAI ",(0,o.jsx)(n.code,{children:"/v1/responses"})]}),"\n",(0,o.jsx)(n.p,{children:"Request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -X POST http://${IP}/v1/responses \\\n    -H \'Content-Type: application/json\' \\\n    -d \'{\n        "model": "\'"${MODEL_NAME}"\'",\n        "input": "Hello",\n        "max_output_tokens": 10\n    }\' | jq\n'})}),"\n",(0,o.jsx)(n.p,{children:"Response:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-json",children:'{\n  "id": "resp_abc127",\n  "created_at": 1781036107,\n  "incomplete_details": {"reason": "max_output_tokens"},\n  "model": "Qwen/Qwen3-VL-32B-Instruct",\n  "object": "response",\n  "output": [\n    {\n      "id": "msg_abc128",\n      "type": "message",\n      "role": "assistant",\n      "status": "completed",\n      "content": [\n        {\n          "type": "output_text",\n          "text": "Hello! How can I help you today?",\n          "annotations": []\n        }\n      ]\n    }\n  ],\n  "status": "incomplete",\n  "max_output_tokens": 10,\n  "usage": {\n    "input_tokens": 9,\n    "output_tokens": 10,\n    "total_tokens": 19\n  }\n}\n'})}),"\n",(0,o.jsxs)(n.h3,{id:"openai-v1embeddings",children:["OpenAI ",(0,o.jsx)(n.code,{children:"/v1/embeddings"})]}),"\n",(0,o.jsxs)(n.p,{children:["This endpoint requires an embedding model deployment (for example ",(0,o.jsx)(n.code,{children:"Qwen/Qwen3-Embedding-0.6B"}
1),"). Chat/instruct models do not expose this route."]}),"\n",(0,o.jsx)(n.p,{children:"Request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'export MODEL_NAME=Qwen/Qwen3-Embedding-0.6B\ncurl -X POST http://${IP}/v1/embeddings \\\n    -H \'Content-Type: application/json\' \\\n    -d \'{\n        "model": "\'"${MODEL_NAME}"\'",\n        "input": "Hello"\n    }\' | jq\n'})}),"\n",(0,o.jsx)(n.p,{children:"Response (embedding vector truncated for readability):"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-json",children:'{\n  "model": "Qwen/Qwen3-Embedding-0.6B",\n  "object": "list",\n  "data": [\n    {\n      "index": 0,\n      "object": "embedding",\n      "embedding": [-0.01350, -0.02152, -0.01368, -0.03032, 0.00941, "..."]\n    }\n  ],\n  "usage": {\n    "prompt_tokens": 2,\n    "total_tokens": 2,\n    "completion_tokens": 0\n  }\n}\n'})}),"\n",(0,o.jsxs)(n.h3,{id:"anthropic-v1messages",children:["Anthropic ",(0,o.jsx)(n.code,{children:"/v1/messages"})]}),"\n",(0,o.jsx)(n.p,{children:"Request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -X POST http://${IP}/v1/messages \\\n    -H \'Content-Type: application/json\' \\\n    -d \'{\n        "model": "\'"${MODEL_NAME}"\'",\n        "messages": [\n            {\n                "role": "user",\n                "content": [\n                    {"type": "text", "text": "Describe this image."},\n                    {"type": "image", "source": {"type": "url", "url": "https://picsum.photos/640/360"}}\n                ]\n            }\n        ],\n        "max_tokens": 10\n    }\' | jq\n'})}),"\n",(0,o.jsx)(n.p,{children:"Response:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-json",children:'{\n  "id": "chatcmpl-abc125",\n  "type": "message",\n  "role": "assistant",\n  "content": [\n    {\n      "type": "text",\n      "text": "This image is a close-up, shallow-focus photograph"\n    }\n  ],\n  "model": "Qwen/Qwen3-VL-32B-Instruct",\n  "stop_reason": "max_tokens",\n  "usage": {\n    "input_tokens": 234,\n    "output_tokens": 10\n  }\n}\n'})}),"\n",(0,o.jsx)(n.p,{children:"Streaming request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -N -X POST http://${IP}/v1/messages \\\n    -H \'Content-Type: application/json\' \\\n    -d \'{\n        "model": "\'"${MODEL_NAME}"\'",\n        "messages": [\n            {\n                "role": "user",\n                "content": [\n                    {"type": "text", "text": "Describe this image."},\n                    {"type": "image", "source": {"type": "url", "url": "https://picsum.photos/640/360"}}\n                ]\n            }\n        ],\n        "max_tokens": 10,\n        "stream": true\n    }\'\n'})}),"\n",(0,o.jsxs)(n.details,{children:["\n",(0,o.jsx)(n.summary,{children:"Streaming response (SSE)"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{children:'event: message_start\ndata: {"type":"message_start","message":{"i
1d":"chatcmpl-abc126","content":[],"model":"Qwen/Qwen3-VL-32B-Instruct","stop_reason":null,"stop_sequence":null,"usage":{"input_tokens":234,"output_tokens":0}}}\n\nevent: content_block_start\ndata: {"type":"content_block_start","content_block":{"type":"text","text":""},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":"This"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" image"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" captures"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" a"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" serene"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" and"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" atmospheric"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" urban"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" landscape"},"index":0}\n\nevent: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":" at"},"index":0}\n\nevent: content_block_stop\ndata: {"type":"content_block_stop","index":0}\n\nevent: message_delta\ndata: {"type":"message_delta","delta":{"stop_reason":"max_tokens"},"usage":{"input_tokens":234,"output_tokens":10}}\n\nevent: message_stop\ndata: {"type":"message_stop"}\n'})}),"\n"]}),"\n",(0,o.jsxs)(n.h3,{id:"vllm-inferencev1generate",children:["vLLM ",(0,o.jsx)(n.code,{children:"/inference/v1/generate"})]}),"\n",(0,o.jsxs)(n.p,{children:["This endpoint requires the model server to be vLLM. Sampling controls must be nested inside a ",(0,o.jsx)(n.code,{children:"sampling_params"})," object rather than placed at the top level."]}),"\n",(0,o.jsx)(n.p,{children:"Request:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-bash",children:'curl -X POST http://${IP}/inference/v1/generate \\\n    -H \'Content-Type: application/json\' \\\n    -d \'{\n        "model": "\'"${MODEL_NAME}"\'",\n        "token_ids": [9906],\n        "sampling_params": {"max_tokens": 10}\n    }\' | jq\n'})}),"\n",(0,o.jsx)(n.p,{children:"Response:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-json",children:'{\n  "request_id": "abc129",\n  "choices": [\n    {\n      "index": 0,\n      "logprobs": null,\n      "finish_reason": "length",\n      "token_ids": [17993, 1894, 7332, 198, 286, 2415, 1140, 259, 4580, 892]\n    }\n  ]\n}\n'})})]})}function p(e={}){const{wrapper:n}={...(0,l.R)(),...e.components};return n?(0,o.jsx)(n,{...e,children:(0,o.jsx)(r,{...e})}):r(e)}},28453(e,n,t){t.d(n,{R:()=>c,x:()=>d});var s=t(96540);const o={},l=s.createContext(o);function c(e){const n=s.useContext(l);return s.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function d(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(o):e.components||o:c(e.components),s.createElement(l.Provider,{value:n},e.children)}}}]);

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.