1"use strict";(globalThis.webpackChunkllm_d_website=globalThis.webpackChunkllm_d_website||[]).push([[40795],{57751(e,n,s){s.r(n),s.d(n,{assets:()=>l,contentTitle:()=>c,default:()=>h,frontMatter:()=>i,metadata:()=>r,toc:()=>a});const r=JSON.parse('{"id":"api-reference/epp-http-headers","title":"EPP HTTP Headers Reference","description":"This document describes the HTTP headers that the Endpoint Picker (EPP) inspects to manage and control inference requests, specifically for flow control, performance management, and request classification.","source":"@site/versioned_docs/version-0.9/api-reference/epp-http-headers.md","sourceDirName":"api-reference","slug":"/api-reference/epp-http-headers","permalink":"/docs/api-reference/epp-http-headers","draft":false,"unlisted":false,"tags":[],"version":"0.9","frontMatter":{},"sidebar":"docsSidebar","previous":{"title":"RPC APIs","permalink":"/docs/api-reference/epp-grpc-apis"},"next":{"title":"Glossary","permalink":"/docs/api-reference/glossary"}}');var t=s(74848),d=s(28453);const i={},c="EPP HTTP Headers Reference",l={},a=[{value:"Request Classification and Flow Control",id:"request-classification-and-flow-control",level:2},{value:"Service Level Objectives (SLOs)",id:"service-level-objectives-slos",level:2},{value:"Deprecated Aliases",id:"deprecated-aliases",level:2},{value:"Response Headers",id:"response-headers",level:2},{value:"Dropped Reason Values",id:"dropped-reason-values",level:3},{value:"Implementation Notes",id:"implementation-notes",level:2}];function o(e){const n={a:"a",code:"code",h1:"h1",h2:"h2",h3:"h3",header:"header",hr:"hr",li:"li",p:"p",strong:"strong",table:"table",tbody:"tbody",td:"td",th:"th",thead:"thead",tr:"tr",ul:"ul",...(0,d.R)(),...e.components};return(0,t.jsxs)(t.Fragment,{children:[(0,t.jsx)(n.header,{children:(0,t.jsx)(n.h1,{id:"epp-http-headers-reference",children:"EPP HTTP Headers Reference"})}),"\n",(0,t.jsxs)(n.p,{children:["This document describes the HTTP headers that the ",(0,t.jsx)(n.a,{href:"../architecture/core/router/epp",children:"Endpoint Picker (EPP)"})," inspects to manage and control inference requests, specifically for flow control, performance management, and request classification."]}),"\n",(0,t.jsx)(n.h2,{id:"request-classification-and-flow-control",children:"Request Classification and Flow Control"}),"\n",(0,t.jsx)(n.p,{children:"These headers allow the EPP to identify the request's goals, group them for fair resource allocation, and handle model-specific targeting."}),"\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n",(0,t.jsxs)(n.table,{children:[(0,t.jsx)(n.thead,{children:(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.th,{children:"Header"}),(0,t.jsx)(n.th,{children:"Description"})]})}),(0,t.jsxs)(n.tbody,{children:[(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-llm-d-inference-objective"})}),(0,t.jsxs)(n.td,{children:["Specifies the name of the ",(0,t.jsx)(n.code,{children:"InferenceObjective"})," resource associated with the request. The EPP uses this to look up the corresponding objective resource in the ",(0,t.jsx)(n.strong,{children:"same namespace as the InferencePool"})," to apply the defined priority and performance goals."]})]}),(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-llm-d-inference-fairness-id"})}),(0,t.jsxs)(n.td,{children:["Provides a unique identifier for grouping requests for fairness-based flow control. Requests with the same ID share capacity according to the fairness policy. If omitted, the EPP defaults to ",(0,t.jsx)(n.code,{children:"default-flow"}),"."]})]}),(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-llm-d-model-name-rewrite"})}),(0,t.jsxs)(n.td,{children:["Specifies the target model name to be used for the request. This is an alternative approach to model name rewriting; while the ",(0,t.jsx)(n.code,{children:"InferenceModelRewrite"})," API provides rule-based rewriting on the server side, this header allows for an explicit, per-request override. When present, the EPP uses this value to override the model name in the request body and for recording model-specific metrics."]})]})]})]}),"\n",(0,t.jsx)(n.h2,{id:"service-level-objectives-slos",children:"Service Level Objectives (SLOs)"}),"\n",(0,t.jsx)(n.p,{children:"These headers are used by admission control and load balancing plugins to make decisions based on latency targets."}),"\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n",(0,t.jsxs)(n.table,{children:[(0,t.jsx)(n.thead,{children:(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.th,{children:"Header"}),(0,t.jsx)(n.th,{children:"Description"})]})}),(0,t.jsxs)(n.tbody,{children:[(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-llm-d-slo-ttft-ms"})}),(0,t.jsx)(n.td,{children:"Specifies the target Time To First Token (TTFT) in milliseconds. Used by plugins to determine if a request can be admitted while meeting the latency goal."})]}),(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-llm-d-slo-tpot-ms"})}),(0,t.jsx)(n.td,{children:"Specifies the target Time Per Output Token (TPOT) in milliseconds. Used for admission control based on predicted or observed token generation latency."})]})]})]}),"\n",(0,t.jsx)(n.h2,{id:"deprecated-aliases",children:"Deprecated Aliases"}),"\n",(0,t.jsxs)(n.p,{children:["llm-d Router v0.9.0 and later accept the previous EPP-managed header names as deprecated read aliases. New integrations and examples should use the canonical ",(0,t.jsx)(n.code,{children:"x-llm-d-*"})," names. Earlier releases expect the previous names, so mixed-version fleets should coordinate the header migration during rolling upgrades. See the ",(0,t.jsx)(n.a,{href:"https://github.com/llm-d/llm-d-router/releases",children:"llm-d Router releases"})," page for available versions."]}),"\n",(0,t.jsxs)(n.p,{children:["If a request sends both a canonical header and its deprecated alias, the canonical ",(0,t.jsx)(n.code,{children:"x-llm-d-*"})," value wins deterministically. No alias removal release or date has been announced; treat aliases as temporary compatibility support and prefer the canonical names for all new clients."]}),"\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n",(0,t.jsxs)(n.table,{children:[(0,t.jsx)(n.thead,{children:(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.th,{children:"Deprecated alias"}),(0,t.jsx)(n.th,{children:"Canonical header"})]})}),(0,t.jsxs)(n.tbody,{children:[(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-gateway-inference-fairness-id"})}),(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-llm-d-inference-fairness-id"})})]}),(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-gateway-inference-objective"})}),(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-llm-d-inference-objective"})})]}),(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-gateway-model-name-rewrite"})}),(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-llm-d-model-name-rewrite"})})]}),(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-slo-ttft-ms"})}),(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-llm-d-slo-ttft-ms"})})]}),(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-slo-tpot-ms"})}),(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-llm-d-slo-tpot-ms"})})]})]})]}),"\n",(0,t.jsxs)(n.p,{children:["This rename applies only to llm-d/EPP-managed user and control headers. ",(0,t.jsx)(n.a,{href:"https://github.com/kubernetes-sigs/gateway-api-inference-extension/tree/main/docs/proposals/004-endpoint-picker-protocol",children:"Gateway API Inference Extension (GAIE) Endpoint Picker Protocol"})," headers, such as ",(0,t.jsx)(n.code,{children:"x-gateway-destination-endpoint*"}),", are unchanged."]}),"\n",(0,t.jsx)(n.h2,{id:"response-headers",children:"Response Headers"}),"\n",(0,t.jsx)(n.p,{children:"These headers are set by the EPP on responses sent back to the client."}),"\n\n\n\n\n\n\n\n\n\n\n\n\n\n",(0,t.jsxs)(n.table,{children:[(0,t.jsx)(n.thead,{children:(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.th,{children:"Header"}),(0,t.jsx)(n.th,{children:"Description"})]})}),(0,t.jsx)(n.tbody,{children:(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"x-llm-d-request-dropped-reason"})}),(0,t.jsxs)(n.td,{children:["Indicates why a request was dropped by flow control. Only present on ",(0,t.jsx)(n.code,{children:"429"})," responses generated by the EPP (not forwarded from a model server). The value uses a two-prefix scheme: ",(0,t.jsx)(n.code,{children:"rejected-*"})," means the request never reached an inference server, ",(0,t.jsx)(n.code,{children:"evicted-*"})," means it was dispatched and then killed. See the table below for possible values."]})]})})]}),"\n",(0,t.jsx)(n.h3,{id:"dropped-reason-values",children:"Dropped Reason Values"}),"\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n",(0,t.jsxs)(n.table,{children:[(0,t.jsx)(n.thead,{children:(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.th,{children:"Value"}),(0,t.jsx)(n.th,{children:"Meaning"})]})}),(0,t.jsxs)(n.tbody,{children:[(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"rejected-saturated"})}),(0,t.jsx)(n.td,{children:"System at capacity, request rejected before queueing."})]}),(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"rejected-ttl-ex
1pired"})}),(0,t.jsx)(n.td,{children:"Request entered the queue but its TTL expired before dispatch."})]}),(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"rejected-context-cancelled"})}),(0,t.jsx)(n.td,{children:"Client disconnected while the request was queued."})]}),(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"evicted"})}),(0,t.jsx)(n.td,{children:"Generic post-dispatch eviction."})]}),(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"evicted-queue-pressure"})}),(0,t.jsx)(n.td,{children:"Evicted after dispatch due to backpressure from new arrivals."})]}),(0,t.jsxs)(n.tr,{children:[(0,t.jsx)(n.td,{children:(0,t.jsx)(n.code,{children:"evicted-priority"})}),(0,t.jsx)(n.td,{children:"Evicted after dispatch, preempted by a higher-priority request."})]})]})]}),"\n",(0,t.jsxs)(n.p,{children:["The two-prefix convention helps consumers decide retry strategy: ",(0,t.jsx)(n.code,{children:"rejected-*"})," means no inference work was done (cheap to retry), ",(0,t.jsx)(n.code,{children:"evicted-*"})," means GPU cycles were consumed (factor into backoff)."]}),"\n",(0,t.jsx)(n.hr,{}),"\n",(0,t.jsx)(n.h2,{id:"implementation-notes",children:"Implementation Notes"}),"\n",(0,t.jsxs)(n.ul,{children:["\n",(0,t.jsxs)(n.li,{children:[(0,t.jsx)(n.strong,{children:"Case Sensitivity:"})," All header lookups are case-insensitive."]}),"\n",(0,t.jsxs)(n.li,{children:[(0,t.jsx)(n.strong,{children:"Source:"})," These values are typically provided as standard HTTP headers in the incoming request."]}),"\n"]})]})}function h(e={}){const{wrapper:n}={...(0,d.R)(),...e.components};return n?(0,t.jsx)(n,{...e,children:(0,t.jsx)(o,{...e})}):o(e)}},28453(e,n,s){s.d(n,{R:()=>i,x:()=>c});var r=s(96540);const t={},d=r.createContext(t);function i(e){const n=r.useContext(d);return r.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function c(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(t):e.components||t:i(e.components),r.createElement(d.Provider,{value:n},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.