1"use strict";(self.webpackChunksemantic_router_docs=self.webpackChunksemantic_router_docs||[]).push([[27405],{55168(e,n,r){r.r(n),r.d(n,{assets:()=>o,contentTitle:()=>s,default:()=>h,frontMatter:()=>d,metadata:()=>t,toc:()=>l});const t=JSON.parse('{"id":"training/mmbert-32k-models","title":"Train Vela Embedding and Reranker","description":"Use Vela Embedding to find relevant documents efficiently, then Vela Reranker","source":"@site/docs/training/mmbert-32k-models.md","sourceDirName":"training","slug":"/training/mmbert-32k-models","permalink":"/docs/training/mmbert-32k-models","draft":false,"unlisted":false,"editUrl":"https://github.com/vllm-project/semantic-router/edit/main/website/docs/training/mmbert-32k-models.md","tags":[],"version":"current","frontMatter":{"title":"Train Vela Embedding and Reranker","sidebar_label":"Embedding and Reranking"},"sidebar":"tutorialSidebar","previous":{"title":"Model Catalog","permalink":"/docs/training/model-catalog"},"next":{"title":"Multimodal Embeddings","permalink":"/docs/training/multimodal-embeddings"}}');var i=r(74848),a=r(28453);const d={title:"Train Vela Embedding and Reranker",sidebar_label:"Embedding and Reranking"},s="Train Vela Embedding and Reranker",o={},l=[{value:"Embedding model: bi-encoder",id:"embedding-model-bi-encoder",level:2},{value:"Reranking model: cross-encoder",id:"reranking-model-cross-encoder",level:2},{value:"Choose depth, dimension, and context",id:"choose-depth-dimension-and-context",level:2},{value:"Prepare a training run",id:"prepare-a-training-run",level:2},{value:"Evaluate the trained checkpoint",id:"evaluate-the-trained-checkpoint",level:2},{value:"Deploy the result",id:"deploy-the-result",level:2},{value:"Earlier mmBERT workflows",id:"earlier-mmbert-workflows",level:2}];function c(e){const n={a:"a",code:"code",h1:"h1",h2:"h2",header:"header",li:"li",p:"p",pre:"pre",strong:"strong",table:"table",tbody:"tbody",td:"td",th:"th",thead:"thead",tr:"tr",ul:"ul",...(0,a.R)(),...e.components};return(0,i.jsxs)(i.Fragment,{children:[(0,i.jsx)(n.header,{children:(0,i.jsx)(n.h1,{id:"train-vela-embedding-and-reranker",children:"Train Vela Embedding and Reranker"})}),"\n",(0,i.jsxs)(n.p,{children:["Use Vela Embedding to find relevant documents efficiently, then Vela Reranker\nto improve the order of a smaller candidate set. Both adapt the shared\n",(0,i.jsx)(n.a,{href:"https://huggingface.co/vllm-sr/Vela-1.0-Encoder-307M",children:"Vela Encoder"}),"\nand support a choice of encoder depth and output dimension."]}),"\n",(0,i.jsxs)(n.p,{children:["To use the published models without training, follow\n",(0,i.jsx)(n.a,{href:"/docs/installation/runtime/embeddings",children:"Embeddings and reranking"}),"."]}),"\n",(0,i.jsx)(n.h2,{id:"embedding-model-bi-encoder",children:"Embedding model: bi-encoder"}),"\n",(0,i.jsx)(n.p,{children:"The embedding model encodes queries and documents independently into normalized\nvectors. Precompute document vectors, then compare a query vector with your\nindex using cosine similarity or a dot product."}),"\n",(0,i.jsx)(n.p,{children:"Train with query-positive pairs and useful negative documents. Add semantic\nsimilarity or paraphrase examples when your application also compares requests,\ngroups related text, or detects repeated questions. Include the languages and\ndomains your index will serve."}),"\n",(0,i.jsx)(n.h2,{id:"reranking-model-cross-encoder",children:"Reranking model: cross-encoder"}),"\n",(0,i.jsx)(n.p,{children:"The reranker reads a query and a candidate document together, then returns a\nrelevance score. Run it on the candidates returned by retrieval."}),"\n",(0,i.jsx)(n.p,{children:"Train with complete candidate lists and relevance judgments. Hard negatives\nfrom your retrieval system help the model distinguish plausible but incorrect\nresults. Keep documents with unknown relevance separate from judged negatives.\nThe score orders candidates; it is not a probability that a document is correct."}),"\n",(0,i.jsx)(n.h2,{id:"choose-depth-dimension-and-context",children:"Choose depth, dimension, and context"}),"\n",(0,i.jsxs)(n.table,{children:[(0,i.jsx)(n.thead,{children:(0,i.jsxs)(n.tr,{children:[(0,i.jsx)(n.th,{children:"Setting"}),(0,i.jsx)(n.th,{children:"Available choices"}),(0,i.jsx)(n.th,{children:"Tradeoff"})]})}),(0,i.jsxs)(n.tbody,{children:[(0,i.jsxs)(n.tr,{children:[(0,i.jsx)(n.td,{children:"Encoder depth"}),(0,i.jsx)(n.td,{children:"3, 6, 11, 22 layers"}),(0,i.jsx)(n.td,{children:"Fewer layers reduce encoder computation"})]}),(0,i.jsxs)(n.tr,{children:[(0,i.jsx)(n.td,{children:"Dimension"}),(0,i.jsx)(n.td,{children:"64, 128, 256, 512, 768"}),(0,i.jsx)(n.td,{children:"Smaller embedding vectors reduce index storage and comparison cost"})]}),(0,i.jsxs)(n.tr,{children:[(0,i.jsx)(n.td,{children:"Input limit"}),(0,i.jsx)(n.td,{children:"Up to 32,768 tokens"}),(0,i.jsx)(n.td,{children:"Longer inputs require more memory and time"})]})]})]}),"\n",(0,i.jsx)(n.p,{children:"Training supervises all 20 depth/dimension combinations. Reranker has a trained\nscoring head for each combination; reducing its dimension does not skip encoder\nlayers. Evaluate the combination you intend to deploy."}),"\n",(0,i.jsx)(n.p,{children:"For embedding, use the same model revision, depth, and dimension for queries\nand indexed documents. Rebuild the index when these settings change. For\nreranking, the input budget covers the query, document, and special tokens\ntogether."}),"\n",(0,i.jsx)(n.h2,{id:"prepare-a-training-run",children:"Prepare a training run"}),"\n",(0,i.jsxs)(n.p,{children:["From a repository checkout, create an isolated environment with the appropriate\nPyTorch build for your accelerator and install the\n",(0,i.jsx)(n.a,{href:"https://github.com/vllm-project/semantic-router/blob/main/src/training/model_embeddings/mmbert_32k/requirements.txt",children:"training dependencies"}),".\nROCm uses PyTorch's ",(0,i.jsx)(n.code,{children:"cuda"})," device name."]}),"\n",(0,i.jsx)(n.p,{children:"Choose one starting point:"}),"\n",(0,i.jsxs)(n.ul,{children:["\n",(0,i.jsxs)(n.li,{children:[(0,i.jsx)(n.strong,{children:"New task:"}
1)," download Vela Encoder and initialize the task from that complete\ncheckpoint."]}),"\n",(0,i.jsxs)(n.li,{children:[(0,i.jsx)(n.strong,{children:"Continue a task:"})," download Vela Embedding or Reranker and select\n",(0,i.jsx)(n.code,{children:'initialization: "continued_task"'}),". This keeps the trained encoder and heads\nwhile starting a new optimizer."]}),"\n",(0,i.jsxs)(n.li,{children:[(0,i.jsx)(n.strong,{children:"Resume an interrupted run:"})," use ",(0,i.jsx)(n.code,{children:"--resume"})," to restore its optimizer and\nprogress."]}),"\n"]}),"\n",(0,i.jsxs)(n.p,{children:["Prepare separate training and development corpora, a sampling plan, and a\ntraining configuration. The\n",(0,i.jsx)(n.a,{href:"https://github.com/vllm-project/semantic-router/tree/main/src/training/model_embeddings/mmbert_32k#train-a-new-task-from-a-standard-base",children:"training workflow reference"}),"\nprovides the JSON formats, supported losses, optional teacher supervision, and\nconfiguration fields."]}),"\n",(0,i.jsxs)(n.p,{children:["The commands below assume those files are ready under ",(0,i.jsx)(n.code,{children:"/data/retrieval"}),".\nSet ",(0,i.jsx)(n.code,{children:"VELA_TRAIN_CONFIG_SHA256"})," to the SHA-256 of your configuration file:"]}),"\n",(0,i.jsx)(n.pre,{children:(0,i.jsx)(n.code,{className:"language-bash",children:'export PYTHONPATH="$PWD"\n\npython -m src.training.model_embeddings.mmbert_32k.newbase_training \\\n --config /data/retrieval/task.json \\\n --config-sha256 "${VELA_TRAIN_CONFIG_SHA256:?Set the configuration SHA-256}" \\\n --output /data/retrieval/run --device cuda\n'})}),"\n",(0,i.jsxs)(n.p,{children:["The configuration selects ",(0,i.jsx)(n.code,{children:'task: "embedding"'})," or ",(0,i.jsx)(n.code,{children:'task: "reranker"'}),", datasets,\ninput budgets, training steps, and evaluation intervals. Start with a small\nbudget and inspect the first development results before extending the run."]}),"\n",(0,i.jsx)(n.h2,{id:"evaluate-the-trained-checkpoint",children:"Evaluate the trained checkpoint"}),"\n",(0,i.jsx)(n.p,{children:"Evaluate a saved checkpoint against the same development corpus used for your\nbaseline:"}),"\n",(0,i.jsx)(n.pre,{children:(0,i.jsx)(n.code,{className:"language-bash",children:"python -m src.training.model_embeddings.mmbert_32k.newbase_scoring \\\n --model /data/retrieval/run/step-100 --task embedding \\\n --known-dev /data/retrieval/validation --split validation \\\n --output /data/retrieval/evaluation --device cpu --token-budget 32768\n"})}),"\n",(0,i.jsxs)(n.p,{children:["Replace the checkpoint path with a step your run saved. For a reranker, set\n",(0,i.jsx)(n.code,{children:"--task reranker"}),". Use development results to choose a checkpoint, then run a\nseparate final comparison on held-out data."]}),"\n",(0,i.jsx)(n.p,{children:"For embedding, compare retrieval, similarity, and multilingual transfer.\nFor reranking, keep candidate lists fixed and compare ranking metrics such as\nnDCG. Measure short and long inputs separately, along with latency and memory\nat each deployed depth/dimension."}),"\n",(0,i.jsx)(n.p,{children:"A full MMTEB result requires its complete selected benchmark and matching\nevaluation protocol. A task subset can diagnose gaps, but cannot establish an\noverall benchmark score or leaderboard rank."}),"\n",(0,i.jsx)(n.h2,{id:"deploy-the-result",children:"Deploy the result"}),"\n",(0,i.jsxs)(n.p,{children:["Use ",(0,i.jsx)(n.a,{href:"/docs/installation/runtime/in-process",children:"local model bindings"})," to select the\ncheckpoint and serving engine. Test the selected depth, dimension, and input\nlimit through ",(0,i.jsx)(n.a,{href:"/docs/installation/runtime/lifecycle-diagnostics",children:"route preview"}),".\nAn ONNX deployment needs graphs exported from the same trained weights."]}),"\n",(0,i.jsx)(n.h2,{id:"earlier-mmbert-workflows",children:"Earlier mmBERT workflows"}),"\n",(0,i.jsxs)(n.p,{children:["The original foundation, embedding, and reranker recipes remain in the\n",(0,i.jsx)(n.a,{href:"https://github.com/vllm-project/semantic-router/tree/main/src/training/model_embeddings/mmbert_32k",children:"workflow README"}),".\nTheir checked-in historical configurations reproduce the earlier mmBERT\nworkflow. Use the Vela starting points above for new Vela task training."]})]})}function h(e={}){const{wrapper:n}={...(0,a.R)(),...e.components};return n?(0,i.jsx)(n,{...e,children:(0,i.jsx)(c,{...e})}):c(e)}},28453(e,n,r){r.d(n,{R:()=>d,x:()=>s});var t=r(96540);const i={},a=t.createContext(i);function d(e){const n=t.useContext(a);return t.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function s(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(i):e.components||i:d(e.components),t.createElement(a.Provider,{value:n},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.