1"use strict";(self.webpackChunkintellidocs=self.webpackChunkintellidocs||[]).push([[7768],{3502:(e,n,t)=>{t.r(n),t.d(n,{assets:()=>r,contentTitle:()=>l,default:()=>c,frontMatter:()=>a,metadata:()=>s,toc:()=>h});var i=t(4848),o=t(8453);const a={sidebar_position:1,title:"Run Gemma Offline in Python",sidebar_label:"Gemma",description:"Use Intelli in Python to run Gemma 2 models offline with Keras NLP, Kaggle setup, Chatbot inputs, and optional RAG document context.",keywords:["intelli python gemma","offline gemma chatbot","gemma 2 keras nlp","python rag chatbot","kaggle gemma setup","intellinode chatbot"]},l="Gemma",s={id:"python/offline-chatbot/gemma",title:"Run Gemma Offline in Python",description:"Use Intelli in Python to run Gemma 2 models offline with Keras NLP, Kaggle setup, Chatbot inputs, and optional RAG document context.",source:"@site/docs/python/offline-chatbot/gemma.md",sourceDirName:"python/offline-chatbot",slug:"/python/offline-chatbot/gemma",permalink:"/docs/python/offline-chatbot/gemma",draft:!1,unlisted:!1,editUrl:"https://github.com/intelligentnode/docs/edit/main/intellidocs/docs/python/offline-chatbot/gemma.md",tags:[],version:"current",sidebarPosition:1,frontMatter:{sidebar_position:1,title:"Run Gemma Offline in Python",sidebar_label:"Gemma",description:"Use Intelli in Python to run Gemma 2 models offline with Keras NLP, Kaggle setup, Chatbot inputs, and optional RAG document context.",keywords:["intelli python gemma","offline gemma chatbot","gemma 2 keras nlp","python rag chatbot","kaggle gemma setup","intellinode chatbot"]},sidebar:"pythonSidebar",previous:{title:"Chat with docs",permalink:"/docs/python/chatbot/docs-chat"},next:{title:"Mistral",permalink:"/docs/python/offline-chatbot/mistral"}},r={},h=[{value:"Setup",id:"setup",level:2},{value:"Initial Setup",id:"initial-setup",level:3},{value:"Installing Dependencies",id:"installing-dependencies",level:3},{value:"Importing the Chatbot",id:"importing-the-chatbot",level:3},{value:"Using the Chatbot",id:"using-the-chatbot",level:2},{value:"Retrieval Augmented Generation (RAG)",id:"retrieval-augmented-generation-rag",level:2},{value:"Setting Up RAG",id:"setting-up-rag",level:3},{value:"Updating the Chatbot with RAG",id:"updating-the-chatbot-with-rag",level:3},{value:"RAG Notes",id:"rag-notes",level:3}];function d(e){const n={a:"a",code:"code",h1:"h1",h2:"h2",h3:"h3",li:"li",ol:"ol",p:"p",pre:"pre",ul:"ul",...(0,o.R)(),...e.components};return(0,i.jsxs)(i.Fragment,{children:[(0,i.jsx)(n.h1,{id:"gemma",children:"Gemma"}),"\n",(0,i.jsx)(n.p,{children:"You can use any of the latest Gemma models offline:"}),"\n",(0,i.jsxs)(n.ul,{children:["\n",(0,i.jsxs)(n.li,{children:[(0,i.jsx)(n.code,{children:"gemma2_9b_en"}),"."]}),"\n",(0,i.jsxs)(n.li,{children:[(0,i.jsx)(n.code,{children:"gemma2_27b_en"}),"."]}),"\n",(0,i.jsxs)(n.li,{children:[(0,i.jsx)(n.code,{children:"gemma2_instruct_9b_en"}),"."]}),"\n",(0,i.jsxs)(n.li,{children:[(0,i.jsx)(n.code,{children:"gemma2_instruct_27b_en"}),"."]}),"\n"]}),"\n",(0,i.jsx)(n.p,{children:"It is recommended to use the instruct versions as they adhere more closely to your commands."}),"\n",(0,i.jsx)(n.h2,{id:"setup",children:"Setup"}),"\n",(0,i.jsx)(n.h3,{id:"initial-setup",children:"Initial Setup"}),"\n",(0,i.jsx)(n.p,{children:"To start, you'll need to download the model from Kaggle. Follow these steps:"}),"\n",(0,i.jsxs)(n.ol,{children:["\n",(0,i.jsx)(n.li,{children:"Create an account on Kaggle."}),"\n",(0,i.jsxs)(n.li,{children:["Go to the ",(0,i.jsx)(n.a,{href:"https://www.kaggle.com/models/keras/gemma2",children:"model page"})," and approve the license."]}),"\n",(0,i.jsx)(n.li,{children:"Generate your access token by clicking on your profile image, then 'Settings', and then the 'Create New Token' button."}),"\n"]}),"\n",(0,i.jsx)(n.p,{children:"These credentials will be used once to download the model. After that, all subsequent steps will run offline."}),"\n",(0,i.jsx)(n.h3,{id:"installing-dependencies",children:"Installing Dependencies"}),"\n",(0,i.jsx)(n.pre,{children:(0,i.jsx)(n.code,{className:"language-python",children:"!pip install keras-nlp\n!pip install --upgrade keras>=3\n!pip install --upgrade intelli\n"})}),"\n",(0,i.jsx)(n.h3,{id:"importing-the-chatbot",children:"Importing the Chatbot"}),"\n",(0,i.jsx)(n.p,{children:"Import the unified offline chatbot from Intellinode:"}),"\n",(0,i.jsx)(n.pre,{children:(0,i.jsx)(n.code,{className:"language-python",children:"from intelli.function.chatbot import Chatbot, ChatProvider\nfrom intelli.model.input.chatbot_input import ChatModelInput\n"})}),"\n",(0,i.jsx)(n.h2,{id:"using-the-chatbot",children:"Using the Chatbot"}),"\n",(0,i.jsx)(n.p,{children:"Set up the model parameters:"}),"\n",(0,i.jsx)(n.pre,{children:(0,i.jsx)(n.code,{className:"language-python",children:'model_params = {\n "model_name": "gemma2_instruct_9b_en",\n "model_params": {\n "KAGGLE_USERNAME": kaggle_user,\n "KAGGLE_KEY": kaggle_key\n }\n}\n'})}),"\n",(0,i.jsx)(n.p,{children:"Initialize the chatbot:"}),"\n",(0,i.jsx)(n.pre,{children:(0,i.jsx)(n.code,{className:"language-python",children:"gemma_bot = Chatbot(provider=ChatProvider.KERAS, options=model_params)\n"})}),"\n",(0,i.jsx)(n.p,{children:"Prepare the input instructions:"}),"\n",(0,i.jsx)(n.pre,{children:(0,i.jsx)(n.code,{className:"language-python",children:'input = ChatModelInput("You are a helpful assistant.")\ninput.max_tokens = 100\ninput.add_user_message("Explain the theory of relativity.")\n'})}),"\n",(0,i.jsx)(n.p,{children:"Execute the chatbot:"}),"\n",(0,i.jsx)(n.pre,{children:(0,i.jsx)(n.code,{className:"language-python",children:"response = gemma_bot.chat(input)\n"})}),"\n",(0,i.jsx)(n.h2,{id:"retrieval-augmented-generation-rag",children:"Retrieval Augmented Generation (RAG)"}),"\n",(0,i.jsx)(n.p,{children:"Intellinode allows you to upload your documents for free and generate a key to provide RAG capabilities to any open-source chatbot, enhancing the model's ability to answer questions using your data."}),"\n",(0,i.jsx)(n.h3,{id:"setting-up-rag",children:"Setting Up RAG"}),"\n",(0,i.jsxs)(n.ol,{children:["\n",(0,i.jsxs)(n.li,{children:["Go to ",(0,i.jsx)(n.a,{href:"https://app.intellinode.ai/",children:"Intellinode cloud"}),"."]}),"\n",(0,i.jsx)(n.li,{children:"Start a new project with the default settings."}),"\n",(0,i.jsx)(n.li,{children:"Upload any PDF (preferred), JSON, word, image, code, or CSV file."}),"\n",(0,i.jsx)(n.li,{children:"After the document is uploaded successfully, copy the provided one key for RAG integration."}),"\n"]}),"\n",(0,i.jsx)(n.h3,{id:"updating-the-chatbot-with-rag",children:"Updating the Chatbot with RAG"}),"\n",(0,i.jsx)(n.p,{children:"Update the chatbot with the RAG key:"}),"\n",(0,i.jsx)(n.pre,{children:(0,i.jsx)(n.code,{className:"language-python",children:"gemma_bot.add_rag({'one_key': '<your-rag-key>'})\n"})}),"\n",(0,i.jsx)(n.p,{children:"Prepare the input instructions:"}),"\n",(0,i.jsx)(n.pre,{children:(0,i.jsx)(n.code,{className:"language-python",children:'input = ChatModelInput("Answer only from the context.")\ninput.max_tokens = 2000\ninput.search_k = 2 # number of returned pages\ninput.add_user_message("What is the red planet?")\n'})}),"\n",(0,i.jsx)(n.p,{children:"Execute"}),"\n",(0,i.jsx)(n.pre,{children:(0,i.jsx)(n.code,{className:"language-python",children:"response = gemma_bot.chat(input)\n"})}),"\n",(0,i.jsx)(n.h3,{id:"rag-notes",children:"RAG Notes"}),"\n",(0,i.jsxs)(n.ol,{children:["\n",(0,i.jsxs)(n.li,{children:["\n",(0,i.jsx)(n.p,{children:"Increase the number of max tokens when increasing the number of searched pages (search_k). The model generates both the input and output with each iteration, If the input contains too many pages and the max tokens is too low, the model may only regenerate the input without leaving space for the output."}),"\n"]}),"\n",(0,i.jsxs)(n.li,{children:["\n",(0,i.jsx)(n.p,{children:"Increasing the max tokens can impact the model's response time and require higher computational resources."}),"\n"]}),"\n",(0,i.jsxs)(n.li,{children:["\n",(0,i.jsx)(n.p,{children:"Increasing the number of returned pages (search_k) improves the accuracy, but requires an increase in the max tokens."}),"\n"]}),"\n",(0,i.jsxs)(n.li,{children:["\n",(0,i.jsxs)(n.p,{children:["Use instructions like ",(0,i.jsx)(n.code,{children:'ChatModelInput("Answer only from the context.")'})," to guide the model to strict responses based on the provided documents."]}),"\n"]}),"\n"]})]})}function c(e={}){const{wrapper:n}={...(0,o.R)(),...e.components};return n?(0,i.jsx)(n,{...e,children:(0,i.jsx)(d,{...e})}):d(e)}},8453:(e,n,t)=>{t.d(n,{R:()=>l,x:()=>s});var i=t(6540);const o={},a=i.createContext(o);function l(e){const n=i.useContext(a);return i.useMemo((function(){return"function"==typeof e?e(n):{...n,...e}}),[n,e])}function s(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(o):e.components||o:l(e.components),i.createElement(a.Provider,{value:n},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.