1"use strict";(globalThis.webpackChunkmy_website=globalThis.webpackChunkmy_website||[]).push([[17755],{23101(e,n,i){i.d(n,{Ay:()=>o,RM:()=>s});var a=i(74848),t=i(28453);const s=[{value:"AutoPQ",id:"autopq",level:2},{value:"Flat vector index + Binary Quantization",id:"flat-vector-index--binary-quantization",level:2},{value:"Binary quantization",id:"binary-quantization",level:3},{value:"OSS LLM integration with <code>generative-anyscale</code>",id:"oss-llm-integration-with-generative-anyscale",level:2},{value:"Python client beta update",id:"python-client-beta-update",level:2},{value:"Performance improvements",id:"performance-improvements",level:2},{value:"Minor changes",id:"minor-changes",level:2},{value:"Summary",id:"summary",level:2}];function r(e){const n={a:"a",admonition:"admonition",code:"code",em:"em",h2:"h2",h3:"h3",img:"img",li:"li",ol:"ol",p:"p",strong:"strong",ul:"ul",...(0,t.R)(),...e.components};return(0,a.jsxs)(a.Fragment,{children:[(0,a.jsxs)(n.p,{children:["Weaviate ",(0,a.jsx)(n.code,{children:"1.23"})," is here!"]}),"\n",(0,a.jsxs)(n.p,{children:["Here are the release \u2b50\ufe0f",(0,a.jsx)(n.em,{children:"highlights"}),"\u2b50\ufe0f relating to this release:"]}),"\n",(0,a.jsx)(n.p,{children:(0,a.jsx)(n.img,{alt:"Weaviate 1.23",src:i(90881).A+"",width:"1801",height:"946"})}),"\n",(0,a.jsxs)(n.ol,{children:["\n",(0,a.jsxs)(n.li,{children:[(0,a.jsx)(n.strong,{children:"AutoPQ"})," - Weaviate now automatically triggers the use of Product Quantization (PQ) for vector indexing. This improves the developer experience for switching on PQ. We've also added auto ",(0,a.jsx)(n.code,{children:"segment"})," size setting."]}),"\n",(0,a.jsxs)(n.li,{children:[(0,a.jsx)(n.strong,{children:"Flat vector index + Binary Quantization"})," - New index type for small collections, such as for multi-tenancy use cases."]}),"\n",(0,a.jsxs)(n.li,{children:[(0,a.jsx)(n.strong,{children:(0,a.jsx)(n.code,{children:"generative-anyscale"})})," - Adds open-source large language model integration."]}),"\n",(0,a.jsxs)(n.li,{children:[(0,a.jsx)(n.strong,{children:"Performance improvements"})," - Mean time to recovery (MMTR) is reduced and automatic resource limiting prevents out-of-memory errors."]}),"\n",(0,a.jsxs)(n.li,{children:[(0,a.jsx)(n.strong,{children:"Python client beta update"})," - Adds ",(0,a.jsx)(n.code,{children:"1.23"})," support and new features."]}),"\n",(0,a.jsxs)(n.li,{children:[(0,a.jsx)(n.strong,{children:"Minor changes"})," - The ",(0,a.jsx)(n.code,{children:"nodes"})," endpoint adds a new ",(0,a.jsx)(n.code,{children:"minimal"})," output default."]}),"\n"]}),"\n",(0,a.jsx)(n.admonition,{title:"Available on WCD",type:"tip",children:(0,a.jsxs)(n.p,{children:[(0,a.jsx)(n.code,{children:"1.23"})," is already available on ",(0,a.jsx)(n.a,{href:"https://console.weaviate.cloud/",children:"Weaviate Cloud"})," - so try it out!"]})}),"\n",(0,a.jsx)(n.p,{children:"For more details, keep scrolling \u2b07\ufe0f!"}),"\n",(0,a.jsx)(n.h2,{id:"autopq",children:"AutoPQ"}),"\n",(0,a.jsxs)(n.p,{children:[(0,a.jsx)(n.img,{alt:"AutoPQ",src:i(94318).A+"#gh-dark-mode-only",width:"1201",height:"562"}),"\n",(0,a.jsx)(n.img,{alt:"AutoPQ",src:i(22058).A+"#gh-light-mode-only",width:"1201",height:"562"})]}),"\n",(0,a.jsxs)(n.p,{children:["Weaviate introduced Product Quantization (PQ) earlier this year. Since then, we've improved how PQ works with your data. In v1.23 we've made it easier to get started. PQ requires a training step. We've heard that the training step was tricky to configure, so we created AutoPQ to take care of the training for you. Just ",(0,a.jsx)(n.a,{href:"https://docs.weaviate.io/weaviate/configuration/compression/pq-compression/#configure-autopq",children:"enable AutoPQ"})," in your system configuration. Then, any time you enable PQ on a new collection, AutoPQ takes care of training and initializes PQ for you."]}),"\n",(0,a.jsxs)(n.p,{children:["We have other improvements too. PQ uses ",(0,a.jsx)(n.a,{href:"https://docs.weaviate.io/weaviate/concepts/vector-quantization/#product-quantization",children:"segments"})," to compress vectors. In this release we have a new algorithm to determine the optimal segment size for your vectors. You can still set the segment size manually, but you shouldn't have to."]}),"\n",(0,a.jsx)(n.p,{children:"Together, AutoPQ and improved segment sizing make using PQ easier than ever."}),"\n",(0,a.jsx)(n.h2,{id:"flat-vector-index--binary-quantization",children:"Flat vector index + Binary Quantization"}),"\n",(0,a.jsxs)(n.p,{children:[(0,a.jsx)(n.img,{alt:"flat-index",src:i(67002).A+"#gh-dark-mode-only",width:"1201",height:"923"}),"\n",(0,a.jsx)(n.img,{alt:"flat-index",src:i(79134).A+"#gh-light-mode-only",width:"1201",height:"923"})]}),"\n",(0,a.jsxs)(n.p,{children:["Weaviate now supports a ",(0,a.jsx)(n.code,{children:"flat"})," vector index type in addition to the existing ",(0,a.jsx)(n.code,{children:"hnsw"})," index."]}),"\n",(0,a.jsxs)(n.p,{children:["As the name suggests, the ",(0,a.jsx)(n.code,{children:"flat"})," index is a single layer of disk-based references to the object vectors. It therefore has a correspondingly small size and minimal memory footprint."]}),"\n",(0,a.jsxs)(n.p,{children:["This index type is particularly useful for multi-tenancy use cases, where each tenant's collection is relatively small, and thus does not need the overhead that comes with building ",(0,a.jsx)(n.code,{children:"hnsw"})," indexes."]}),"\n",(0,a.jsxs)(n.p,{children:["The ",(0,a.jsx)(n.code,{children:"flat"})," index can be optionally combined with binary quantization (BQ)."]}),"\n",(0,a.jsx)(n.h3,{id:"binary-quantization",children:"Binary quantization"}),"\n",(0,a.jsxs)(n.p,{children:["Binary quantization (BQ) compression is available for the ",(0,a.jsx)(n.code,{children:"flat"})," index type to speed up vector search."]}
1),"\n",(0,a.jsx)(n.p,{children:"BQ works by converting each vector to a binary representation, such as consisting of N dimensions of signs. This binary representation is then used for distance calculations, instead of the original vector."}),"\n",(0,a.jsxs)(n.p,{children:["Weaviate deals with any loss in vector similarity accuracy by conditionally over-fetching and then re-scoring the results. Anecdotally, we have seen encouraging recall with Cohere's V3 models (e.g. ",(0,a.jsx)(n.code,{children:"embed-multilingual-v3.0"})," or ",(0,a.jsx)(n.code,{children:"embed-english-v3.0"}),"), and OpenAI's ",(0,a.jsx)(n.code,{children:"ada-002"})," model with BQ enabled."]}),"\n",(0,a.jsx)(n.p,{children:"We expect that BQ will generally work better for vectors with higher dimensions. We advise you to test BQ with your own data and preferred vectorizer to determine if it is suitable for your use case."}),"\n",(0,a.jsx)(n.p,{children:"When BQ is enabled, a vector cache can be used to improve query performance by storing the quantized vectors of the most recently used data objects. Note that it must be balanced with memory usage considerations."}),"\n",(0,a.jsxs)(n.ul,{children:["\n",(0,a.jsxs)(n.li,{children:["Read more about the ",(0,a.jsx)(n.code,{children:"flat"})," index ",(0,a.jsx)(n.a,{href:"https://docs.weaviate.io/weaviate/concepts/vector-index#flat-index",children:"here"}),"."]}),"\n"]}),"\n",(0,a.jsxs)(n.h2,{id:"oss-llm-integration-with-generative-anyscale",children:["OSS LLM integration with ",(0,a.jsx)(n.code,{children:"generative-anyscale"})]}),"\n",(0,a.jsx)(n.p,{children:(0,a.jsx)(n.img,{alt:"Weaviate 1.23",src:i(87826).A+"",width:"1200",height:"675"})}),"\n",(0,a.jsxs)(n.p,{children:["With the ",(0,a.jsx)(n.code,{children:"1.23"})," release, it is easier to use Weaviate with many open-source large language models (LLMs) such as Llama2-70b, CodeLlama-34b or Mistral-7B-Instruct. This is made possible by the ",(0,a.jsx)(n.a,{href:"https://docs.weaviate.io/weaviate/model-providers/anyscale/generative",children:"generative-anyscale"})," module."]}),"\n",(0,a.jsxs)(n.p,{children:["This module integrates Weaviate with the ",(0,a.jsx)(n.a,{href:"https://www.anyscale.com/",children:"Anyscale"})," service, which provides a hosted inference service for large language models. This allows Weaviate users to perform retrieval augmented generation (RAG) with open-source LLMs, without having to worry about the infrastructure required to run these models."]}),"\n",(0,a.jsx)(n.p,{children:"Currently, these models are supported:"}),"\n",(0,a.jsxs)(n.ul,{children:["\n",(0,a.jsx)(n.li,{children:(0,a.jsx)(n.code,{children:"meta-llama/Llama-2-70b-chat-hf"})}),"\n",(0,a.jsx)(n.li,{children:(0,a.jsx)(n.code,{children:"meta-llama/Llama-2-13b-chat-hf"})}),"\n",(0,a.jsx)(n.li,{children:(0,a.jsx)(n.code,{children:"meta-llama/Llama-2-7b-chat-hf"})}),"\n",(0,a.jsx)(n.li,{children:(0,a.jsx)(n.code,{children:"codellama/CodeLlama-34b-Instruct-hf"})}),"\n",(0,a.jsx)(n.li,{children:(0,a.jsx)(n.code,{children:"mistralai/Mistral-7B-Instruct-v0.1"})}),"\n",(0,a.jsx)(n.li,{children:(0,a.jsx)(n.code,{children:"mistralai/Mixtral-8x7B-Instruct-v0.1"})}),"\n"]}),"\n",(0,a.jsxs)(n.p,{children:["If you have used any of the other ",(0,a.jsx)(n.code,{children:"generative"})," modules in Weaviate, the usage pattern is identical. Make sure you supply your Anyscale API key to Weaviate, and enjoy using these models!"]}),"\n",(0,a.jsxs)(n.ul,{children:["\n",(0,a.jsxs)(n.li,{children:["Read more about the ",(0,a.jsx)(n.code,{children:"generative-anyscale"})," module ",(0,a.jsx)(n.a,{href:"https://docs.weaviate.io/weaviate/model-providers/anyscale/generative",children:"here"}),"."]}),"\n"]}),"\n",(0,a.jsx)(n.h2,{id:"python-client-beta-update",children:"Python client beta update"}),"\n",(0,a.jsxs)(n.p,{children:["The Weaviate Python client has been updated to support the new ",(0,a.jsx)(n.code,{children:"1.23"})," features. This release also includes additional syntax changes to make the client more intuitive."]}),"\n",(0,a.jsxs)(n.p,{children:["This ",(0,a.jsx)(n.code,{children:"4.4b3"})," beta release is designed to be used with Weaviate ",(0,a.jsx)(n.code,{children:"1.23"}),". The nature of gRPC means that many changes are coupled between the server and the client. If you upgrade Weaviate to ",(0,a.jsx)(n.code,{children:"1.23"}),", please also update the Python client to use them together."]}),"\n",(0,a.jsx)(n.p,{children:"Some of the changes in this release include:"}),"\n",(0,a.jsxs)(n.ul,{children:["\n",(0,a.jsxs)(n.li,{children:["\n",(0,a.jsxs)(n.p,{children:[(0,a.jsx)(n.code,{children:"metadata"})," based filtering was added."]}),"\n"]}),"\n",(0,a.jsxs)(n.li,{children:["\n",(0,a.jsxs)(n.p,{children:["Raw GraphQL queries can be performed through ",(0,a.jsx)(n.code,{children:"client.graphql_raw_query()"}),"."]}),"\n"]}),"\n",(0,a.jsxs)(n.li,{children:["\n",(0,a.jsxs)(n.p,{children:["Backups for individual collections (",(0,a.jsx)(n.code,{children:"client.collection.backup"}),") or the entire instance (",(0,a.jsx)(n.code,{children:"client.backup"}),")."]}),"\n"]}),"\n",(0,a.jsxs)(n.li,{children:["\n",(0,a.jsxs)(n.p,{children:[(0,a.jsx)(n.code,{children:"references"}
1)," are their own parameters inputs where applicable, and returned under their own attributes. For example:"]}),"\n",(0,a.jsxs)(n.ul,{children:["\n",(0,a.jsxs)(n.li,{children:["The ",(0,a.jsx)(n.code,{children:"client.collections.create"})," function includes a ",(0,a.jsx)(n.code,{children:"references"})," parameter."]}),"\n",(0,a.jsxs)(n.li,{children:["Returned query results include a ",(0,a.jsx)(n.code,{children:"references"})," attribute where cross-references were queried."]}),"\n"]}),"\n"]}),"\n",(0,a.jsxs)(n.li,{children:["\n",(0,a.jsxs)(n.p,{children:["Native ",(0,a.jsx)(n.code,{children:"datetime"})," objects are used where applicable, such as for time-based ",(0,a.jsx)(n.code,{children:"metadata"})," attributes or date properties."]}),"\n"]}),"\n",(0,a.jsxs)(n.li,{children:["\n",(0,a.jsxs)(n.p,{children:["Read more about the ",(0,a.jsx)(n.code,{children:"Python"})," client ",(0,a.jsx)(n.code,{children:"v4"})," ",(0,a.jsx)(n.a,{href:"https://docs.weaviate.io/weaviate/client-libraries/python",children:"here"}),"."]}),"\n"]}),"\n"]}),"\n",(0,a.jsx)(n.h2,{id:"performance-improvements",children:"Performance improvements"}),"\n",(0,a.jsxs)(n.ul,{children:["\n",(0,a.jsxs)(n.li,{children:["\n",(0,a.jsxs)(n.p,{children:[(0,a.jsx)(n.a,{href:"https://docs.weaviate.io/weaviate/concepts/data#lazy-shard-loading",children:"Lazy shard loading"})," allows you to start working with your data sooner. After a restart, shards load in the background. If the shard you want to query is already loaded, you can get your results right away. If the shard is not loaded yet, Weaviate prioritizes loading that shard and returns a response when it is ready."]}),"\n"]}),"\n",(0,a.jsxs)(n.li,{children:["\n",(0,a.jsxs)(n.p,{children:["You can now enable an option to ",(0,a.jsx)(n.a,{href:"https://docs.weaviate.io/weaviate/concepts/resources#limit-available-resources",children:"auto-limit available resources"})," in Weaviate. In applicable systems, you can set the ",(0,a.jsx)(n.code,{children:"LIMIT_RESOURCES"})," ",(0,a.jsx)(n.a,{href:"https://docs.weaviate.io/weaviate/config-refs/env-vars",children:"environment variable"}),"."]}),"\n"]}),"\n"]}),"\n",(0,a.jsx)(n.h2,{id:"minor-changes",children:"Minor changes"}),"\n",(0,a.jsxs)(n.p,{children:["The ",(0,a.jsxs)(n.a,{href:"https://docs.weaviate.io/weaviate/config-refs/nodes",children:[(0,a.jsx)(n.code,{children:"nodes"})," endpoint"]})," can be used to output information about the nodes in your cluster."]}),"\n",(0,a.jsxs)(n.p,{children:["This endpoint has been updated with a new ",(0,a.jsx)(n.code,{children:"output"})," parameter that has a ",(0,a.jsx)(n.code,{children:"minimal"})," default. This is useful for those of you with many shards or tenants, as it reduces the amount of data returned by the endpoint."]}),"\n",(0,a.jsx)(n.h2,{id:"summary",children:"Summary"}),"\n",(0,a.jsxs)(n.p,{children:["That's all from us - we hope you enjoy the new features and improvements in Weaviate ",(0,a.jsx)(n.code,{children:"1.23"}),". This release is already available on ",(0,a.jsx)(n.a,{href:"https://console.weaviate.cloud/",children:"WCD"}),". So you can try it out yourself on a free sandbox, or by upgrading!"]}),"\n",(0,a.jsx)(n.p,{children:"Thanks for reading, and see you next time \ud83d\udc4b!"})]})}function o(e={}){const{wrapper:n}={...(0,t.R)(),...e.components};return n?(0,a.jsx)(n,{...e,children:(0,a.jsx)(r,{...e})}):r(e)}},32352(e,n,i){i.r(n),i.d(n,{assets:()=>c,contentTitle:()=>d,default:()=>p,frontMatter:()=>l,metadata:()=>a,toc:()=>h});var a=i(3187),t=i(74848),s=i(28453),r=i(23101),o=i(15768);const l={title:"Weaviate 1.23 Release",slug:"weaviate-1-23-release",authors:["jp","dave"],date:new Date("2023-12-19T00:00:00.000Z"),image:"./img/hero.png",tags:["release","engineering"],description:"Weaviate 1.23 released with AutoPQ, flat indexing + Binary Quantization, OSS LLM support through Anyscale, and more!"},d=void 0,c={image:i(64844).A,authorsImageUrls:[void 0,void 0]},h=[...r.RM,...o.RM];function u(e){return(0,t.jsxs)(t.Fragment,{children:[(0,t.jsx)(r.Ay,{}),"\n","\n",(0,t.jsx)(o.Ay,{})]})}function p(e={}){const{wrapper:n}={...(0,s.R)(),...e.components};return n?(0,t.jsx)(n,{...e,children:(0,t.jsx)(u,{...e})}):u()}},64844(e,n,i){i.d(n,{A:()=>a});const a=i.p+"assets/images/hero-279acc31fc0045c9790ebb9d590b635a.png"},87826(e,n,i){i.d(n,{A:()=>a});const a=i.p+"assets/images/generative-anyscale-91c38aa4ecb3b4934a089fe06d55fbc8.png"},90881(e,n,i){i.d(n,{A:()=>a});const a=i.p+"assets/images/hero-279acc31fc0045c9790ebb9d590b635a.png"},67002(e,n,i){i.d(n,{A:()=>a});const a=i.p+"assets/images/weaviate-release-1-23-flat-index-8a7b9d445068c93a5b6ed6fcd5f8e4a3.png"},79134(e,n,i){i.d(n,{A:()=>a});const a=i.p+"assets/images/weaviate-release-1-23-flat-index_1-035c619baa959beddcfdb09683b650be.png"},94318(e,n,i){i.d(n,{A:()=>a});const a=i.p+"assets/images/weaviate-release-1-23-lazy-1f8d6f106a00097706ce62029a507ef1.png"},22058(e,n,i){i.d(n,{A:()=>a});const a=i.p+"assets/images/weaviate-release-1-23-lazy_1-d02a6e4252e96c985ed230fbbe116308.png"},3187(e){e.exports=JSON.parse('{"permalink":"/blog/weaviate-1-23-release","editUrl":"https://github.com/weaviate/weaviate-io/tree/main/blog/2023-12-19-weaviate-1-23-release/index.mdx","source":"@site/blog/2023-12-19-weaviate-1-23-release/index.mdx","title":"Weaviate 1.23 Release","description":"Weaviate 1.23 released with AutoPQ, flat indexing + Binary Quantization, OSS LLM support through Anyscale, and more!","date":"2023-12-19T00:00:00.000Z","tags":[{"inline":true,"label":"release","permalink":"/blog/tags/release"},{"inline":true,"label":"engineering","permalink":"/blog/tags/engineering"}],"readingTime":0.08,"hasTruncateMarker":false,"authors":[{"name":"Joon-Pil (JP) Hwang","title":"Educator","url":"https://github.com/databyjp","imageURL":"/img/people/icon/jp.jpg","key":"jp","page":null},{"name":"Dave Cuthbert","title":"Technical Writer","url":"https://github.com/daveatweaviate","imageURL":"/img/people/icon/dave.jpg","key":"dave","page":null}],"frontMatter":{"title":"Weaviate 1.23 Release","slug":"weaviate-1-23-release","authors":["jp","dave"],"date":"2023-12-19T00:00:00.000Z","image":"./img/hero.png","tags":["release","engineering"],"description":"Weaviate 1.23 released with AutoPQ, flat indexing + Binary Quantization, OSS LLM support through Anyscale, and more!"},"unlisted":false,"prevItem":{"title":"Weaviate 2023 Recap","permalink":"/blog/2023-recap"},"nextItem":{"title":"Multimodal Retrieval-Augmented Generation (RAG)","permalink":"/blog/multimodal-RAG"}}')}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.