1"use strict";(self.webpackChunkwebsite=self.webpackChunkwebsite||[]).push([["112784"],{507602(e,n,i){i.r(n),i.d(n,{metadata:()=>s,default:()=>m,frontMatter:()=>r,contentTitle:()=>a,toc:()=>c,assets:()=>o});var s=JSON.parse('{"id":"features/embeddings/index","title":"Embedding Datasets","description":"Learn how to define, or augment existing datasets with embedding column(s).","source":"@site/versioned_docs/version-1.11.x/features/embeddings/index.md","sourceDirName":"features/embeddings","slug":"/features/embeddings/","permalink":"/docs/v1.11/features/embeddings/","draft":false,"unlisted":false,"editUrl":"https://github.com/spiceai/docs/edit/trunk/website/versioned_docs/version-1.11.x/features/embeddings/index.md","tags":[],"version":"1.11.x","sidebarPosition":9,"frontMatter":{"title":"Embedding Datasets","sidebar_label":"Embedding Datasets","description":"Learn how to define, or augment existing datasets with embedding column(s).","sidebar_position":9,"pagination_prev":null,"pagination_next":null},"sidebar":"docs"}'),d=i(474848),t=i(28453);let r={title:"Embedding Datasets",sidebar_label:"Embedding Datasets",description:"Learn how to define, or augment existing datasets with embedding column(s).",sidebar_position:9,pagination_prev:null,pagination_next:null},a,o={},c=[{value:"Overview",id:"overview",level:2},{value:"Configuring Embedding Models",id:"configuring-embedding-models",level:2},{value:"Vector Searches",id:"vector-searches",level:2},{value:"Generating Embeddings in Queries",id:"generating-embeddings-in-queries",level:2}];function l(e){let n={a:"a",code:"code",h2:"h2",li:"li",ol:"ol",p:"p",pre:"pre",strong:"strong",...(0,t.R)(),...e.components};return(0,d.jsxs)(d.Fragment,{children:[(0,d.jsx)(n.p,{children:"Learn how to define and augment datasets with embedding columns for advanced search capabilities."}),"\n",(0,d.jsx)(n.h2,{id:"overview",children:"Overview"}),"\n",(0,d.jsx)(n.p,{children:"Spice provides three distinct methods for handling embedding columns in datasets:"}),"\n",(0,d.jsxs)(n.ol,{children:["\n",(0,d.jsxs)(n.li,{children:[(0,d.jsx)(n.strong,{children:(0,d.jsx)(n.a,{href:"../components/embeddings#jit-embeddings",children:"Just-in-Time (JIT) Embeddings"})}),": Dynamically computes embeddings, on-demand, during query execution, without precomputing data."]}),"\n",(0,d.jsxs)(n.li,{children:[(0,d.jsx)(n.strong,{children:(0,d.jsx)(n.a,{href:"../components/embeddings#accelerated-embeddings",children:"Accelerated Embeddings"})}),": Precomputes embeddings by transforming and augmenting the source dataset for faster query and search performance."]}),"\n",(0,d.jsxs)(n.li,{children:[(0,d.jsx)(n.strong,{children:(0,d.jsx)(n.a,{href:"../components/embeddings#passthrough-embeddings",children:"Passthrough Embeddings"})}),": Uses pre-existing embeddings directly from the underlying source datasets, bypassing any additional computation."]}),"\n"]}),"\n",(0,d.jsx)(n.h2,{id:"configuring-embedding-models",children:"Configuring Embedding Models"}),"\n",(0,d.jsxs)(n.p,{children:["Before configuring dataset embeddings, define the embedding models in the ",(0,d.jsx)(n.code,{children:"spicepod.yaml"}),". For example:"]}),"\n",(0,d.jsx)(n.pre,{children:(0,d.jsx)(n.code,{className:"language-yaml",children:"embeddings:\n - name: local_embedding_model\n from: huggingface:huggingface.co/sentence-transformers/all-MiniLM-L6-v2\n\n - from: openai\n name: remote_service\n params:\n openai_api_key: ${ secrets:SPICE_OPENAI_API_KEY }\n"})}),"\n",(0,d.jsxs)(n.p,{children:["See ",(0,d.jsx)(n.a,{href:"../components/embeddings",children:"Embedding components"})," for more information on embedding models."]}),"\n",(0,d.jsx)(n.h2,{id:"vector-searches",children:"Vector Searches"}),"\n",(0,d.jsx)(n.p,{children:"Spice supports complex searches by utilizing embeddings. Both local and remote embedding models can be used for vector searches."}),"\n",(0,d.jsx)(n.p,{children:"To run a vector search, embeddings must be defined for the relevant columns in your dataset. Once configured, similarity searches can be performed using the defined embeddings."}),"\n",(0,d.jsxs)(n.p,{children:["For detailed instructions and examples on running vector searches, refer to the ",(0,d.jsx)(n.a,{href:"search/vector-search",children:"Vector-Based Search documentation"}),"."]}),"\n",(0,d.jsx)(n.h2,{id:"generating-embeddings-in-queries",children:"Generating Embeddings in Queries"}),"\n",(0,d.jsxs)(n.p,{children:["The ",(0,d.jsxs)(n.a,{href:"../reference/sql/scalar_functions#embed",children:[(0,d.jsx)(n.code,{children:"embed()"})," scalar function"]})," generates embeddings directly within SQL queries. This function can process both single text str
1ings and arrays of text, making it useful for ad-hoc embedding generation and comparison operations."]})]})}function m(e={}){let{wrapper:n}={...(0,t.R)(),...e.components};return n?(0,d.jsx)(n,{...e,children:(0,d.jsx)(l,{...e})}):l(e)}},28453(e,n,i){i.d(n,{R:()=>r,x:()=>a});var s=i(296540);let d={},t=s.createContext(d);function r(e){let n=s.useContext(t);return s.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function a(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(d):e.components||d:r(e.components),s.createElement(t.Provider,{value:n},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.