1"use strict";(globalThis.webpackChunkragflow_docs=globalThis.webpackChunkragflow_docs||[]).push([[5299],{16432(e,t,n){n.r(t),n.d(t,{assets:()=>o,contentTitle:()=>s,default:()=>h,frontMatter:()=>l,metadata:()=>i,toc:()=>c});const i=JSON.parse('{"id":"guides/knowledge_compilation/runtime_configuration","title":"Knowledge Compilation Runtime Configuration","description":"Knowledge compilation runtime parameters can be configured with environment","source":"@site/docs/guides/knowledge_compilation/runtime_configuration.md","sourceDirName":"guides/knowledge_compilation","slug":"/knowledge_compilation/runtime_configuration","permalink":"/docs/knowledge_compilation/runtime_configuration","draft":false,"unlisted":false,"editUrl":"https://github.com/infiniflow/ragflow/tree/main/docs/guides/knowledge_compilation/runtime_configuration.md","tags":[],"version":"current","sidebarPosition":6,"frontMatter":{"sidebar_position":6,"title":"Knowledge Compilation Runtime Configuration","sidebar_label":"Runtime Configuration","slug":"/knowledge_compilation/runtime_configuration","sidebar_custom_props":{"categoryIcon":"LucideWandSparkles"}},"sidebar":"tutorialSidebar","previous":{"title":"FAQ","permalink":"/docs/knowledge_compilation/faq"},"next":{"title":"Administrator Guides","permalink":"/docs/category/administrator-guides"}}');var r=n(74848),d=n(28453);const l={sidebar_position:6,title:"Knowledge Compilation Runtime Configuration",sidebar_label:"Runtime Configuration",slug:"/knowledge_compilation/runtime_configuration",sidebar_custom_props:{categoryIcon:"LucideWandSparkles"}},s="Knowledge Compilation Runtime Configuration",o={},c=[{value:"Wiki Configuration",id:"wiki-configuration",level:2},{value:"Structure Compile Configuration",id:"structure-compile-configuration",level:2},{value:"LLM Pool Rate-Limit Handling",id:"llm-pool-rate-limit-handling",level:2},{value:"Example",id:"example",level:2}];function a(e){const t={code:"code",h1:"h1",h2:"h2",header:"header",p:"p",pre:"pre",table:"table",tbody:"tbody",td:"td",th:"th",thead:"thead",tr:"tr",...(0,d.R)(),...e.components};return(0,r.jsxs)(r.Fragment,{children:[(0,r.jsx)(t.header,{children:(0,r.jsx)(t.h1,{id:"knowledge-compilation-runtime-configuration",children:"Knowledge Compilation Runtime Configuration"})}),"\n",(0,r.jsxs)(t.p,{children:["Knowledge compilation runtime parameters can be configured with environment\nvariables. In the Docker deployment, set them in ",(0,r.jsx)(t.code,{children:"docker/.env"})," and restart the\nRAGFlow service for the changes to take effect."]}),"\n",(0,r.jsx)(t.p,{children:"The values below are the defaults. If an environment variable is not set, its\ndefault value is used. Invalid values are replaced with the default value.\nValues below a configured minimum or above a configured maximum are clamped to\nthe valid range and logged."}),"\n",(0,r.jsx)(t.h2,{id:"wiki-configuration",children:"Wiki Configuration"}),"\n",(0,r.jsxs)(t.table,{children:[(0,r.jsx)(t.thead,{children:(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.th,{children:"Environment variable"}),(0,r.jsx)(t.th,{style:{textAlign:"right"},children:"Default"}),(0,r.jsx)(t.th,{children:"Description"})]})}),(0,r.jsxs)(t.tbody,{children:[(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"WIKI_MAP_LLM_POOL_SIZE"})}),(0,r.jsx)(t.td,{style:{textAlign:"right"},children:(0,r.jsx)(t.code,{children:"20"})}),(0,r.jsx)(t.td,{children:"Maximum number of concurrent LLM calls allowed by the Wiki task's shared LLM pool."})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"WIKI_MAP_MAX_PENDING"})}),(0,r.jsx)(t.td,{style:{textAlign:"right"},children:(0,r.jsx)(t.code,{children:"25"})}),(0,r.jsxs)(t.td,{children:["Maximum number of active and waiting calls admitted by the Wiki LLM pool. The effective value is never lower than ",(0,r.jsx)(t.code,{children:"WIKI_MAP_LLM_POOL_SIZE"}),"."]})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"WIKI_REFINE_WORKERS"})}),(0,r.jsx)(t.td,{style:{textAlign:"right"},children:(0,r.jsx)(t.code,{children:"4"})}),(0,r.jsx)(t.td,{children:"Number of page-refinement workers. Actual LLM concurrency is still limited by the shared pool."})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"WIKI_MAP_WORKERS"})}),(0,r.jsx)(t.td,{style:{textAlign:"right"},children:(0,r.jsx)(t.code,{children:"20"})}),(0,r.jsx)(t.td,{children:"Default worker count used by direct Wiki MAP calls."})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"WIKI_MAP_TIMEOUT"})}),(0,r.jsxs)(t.td,{style:{textAlign:"right"},children:[(0,r.jsx)(t.code,{children:"600"})," seconds"]}),(0,r.jsx)(t.td,{children:"Timeout for one Wiki MAP extraction call."})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"WIKI_PLAN_TIMEOUT"})}),(0,r.jsxs)(t.td,{style:{textAlign:"right"},children:[(0,r.jsx)(t.code,{children:"600"})," seconds"]}),(0,r.jsx)(t.td,{children:"Timeout for Wiki PLAN calls, including page planning and MAYBE resolution."})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"WIKI_REFINE_TIMEOUT"})}),(0,r.jsxs)(t.td,{style:{textAlign:"right"},children:[(0,r.jsx)(t.code,{children:"300"})," seconds"]}),(0,r.jsx)(t.td,{children:"Timeout for one Wiki page-writing call."})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"WIKI_MERGE_TIMEOUT"})}),(0,r.jsxs)(t.td,{style:{textAlign:"right"},children:[(0,r.jsx)(t.code,{children:"600"})," seconds"]}),(0,r.jsx)(t.td,{children:"Timeout for merging an existing Wiki page with newly generated content."})]})]})]}),"\n",(0,r.jsxs)(t.p,{children:[(0,r.jsx)(t.code,{children:"WIKI_MAP_LLM_POOL_SIZE"})," is a hard ceiling for the task. The pool can reduce\nthe effective concurrency for a model after rate limiting, so setting this\nvalue does not force every provider to receive that many concurrent requests."]}),"\n",(0,r.jsx)(t.h2,{id:"structure-compile-configuration",children:"Structure Compile Configuration"}),"\n",(0,r.jsxs)(t.table,{children:[(0,r.jsx)(t.thead,{children:(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.th,{children:"Environment variable"}),(0,r.jsx)(t.th,{style:{textAlign:"right"},children:"Default"}),(0,r.jsx)(t.th,{children:"Description"})]})}),(0,r.jsxs)(t.tbody,{children:[(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"DOC_STRUCTURE_LLM_POOL_SIZE"})}),(0,r.jsx)(t.td,{style:{textAlign:"right"},children:(0,r.jsx)(t.code,{children:"20"})}),(0,r.jsx)(t.td,{children:"Maximum number of concurrent LLM calls for a document Structure Compile task."})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"DOC_STRUCTURE_COMPILE_MAX_IN_FLIGHT"})}),(0,r.jsx)(t.td,{style:{textAlign:"right"},children:(0,r.jsx)(t.code,{children:"15"})}),(0,r.jsx)(t.td,{children:"Maximum number of structure batch/template operations in flight."})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"DOC_STRUCTURE_COMPILE_BATCH_CHUNKS"})}),(0,r.jsx)(t.td,{style:{textAlign:"right"},children:(0,r.jsx)(t.code,{children:"4"})}),(0,r.jsx)(t.td,{children:"Number of source chunks passed to one outer Structure Compile batch."})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"STRUCTURE_CONTEXT_FRACTION"})}),(0,r.jsx)(t.td,{style:{textAlign:"right"},children:(0,r.jsx)(t.code,{children:"0.5"})}),(0,r.jsxs)(t.td,{children:["Fraction of the model context window used when packing structure batches, clamped to ",(0,r.jsx)(t.code,{children:"(0, 1]"}),"."]})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"STRUCTURE_DEFAULT_CONTEXT"})}),(0,r.jsxs)(t.td,{style:{textAlign:"right"},children:[(0,r.jsx)(t.code,{children:"100000"})," tokens"]}),(0,r.jsx)(t.td,{children:"Fallback model context size when the model does not provide one."})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"KNOWLEDGE_GRAPH_CONTEXT_FRACTION"})}),(0,r.jsx)(t.td,{style:{textAlign:"right"},children:(0,r.jsx)(t.code,{children:"0.1"})}),(0,r.jsxs)(t.td,{children:["Fraction of the model context window used for Knowledge Graph batches, clamped to ",(0,r.jsx)(t.code,{children:"(0, 1]"}),"."]})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"KNOWLEDGE_GRAPH_MIN_BATCH_TOKENS"})}),(0,r.jsxs)(t.td,{style:{textAlign:"right"},children:[(0,r.jsx)(t.code,{children:"2048"})," tokens"]}),(0,r.jsx)(t.td,{children:"Minimum Knowledge Graph batch size."})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"KNOWLEDGE_GRAPH_MAX_BATCH_TOKENS"})}),(0,r.jsxs)(t.td,{style:{textAlign:"right"},children:[(0,r.jsx)(t.code,{children:"4096"})," tokens"]}),(0,r.jsxs)(t.td,{children:["Maximum Knowledge Graph batch size. The effective value is never lower than ",(0,r.jsx)(t.code,{children:"KNOWLEDGE_GRAPH_MIN_BATCH_TOKENS"}),"."]})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"STRUCTURE_CHAIN_CORRECTION_TIMEOUT_S"})}),(0,r.jsxs)(t.td,{style:{textAlign:"right"},children:[(0,r.jsx)(t.code,{children:"120"})," seconds"]}),(0,r.jsx)(t.td,{children:"Time limit for the Structure Compile chain-correction LLM step."})]})]})]}),"\n",(0,r.jsx)(t.p,{children:"The regular Structure Compile entity/relation extraction path does not define\nan independent application-level timeout. Its request timeout is determined by\nthe configured LLM provider/client."}),"\n",(0,r.jsx)(t.h2,{id:"llm-pool-rate-limit-handling",children:"LLM Pool Rate-Limit Handling"}),"\n",(0,r.jsx)(t.p,{children:"These variables apply to the shared adaptive LLM pool:"}),"\n",(0,r.jsxs)(t.table,{children:[(0,r.jsx)(t.thead,{children:(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.th,{children:"Environment variable"}),(0,r.jsx)(t.th,{style:{textAlign:"right"},children:"Default"}),(0,r.jsx)(t.th,{children:"Description"})]})}),(0,r.jsxs)(t.tbody,{children:[(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"LLM_POOL_RATE_LIMIT_RETRIES"})}),(0,r.jsx)(t.td,{style:{textAlign:"right"},children:(0,r.jsx)(t.code,{children:"3"})}),(0,r.jsx)(t.td,{children:"Number of retries after a rate-limit response."})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"LLM_POOL_RATE_LIMIT_RETRY_BASE_DELAY"})}),(0,r.jsxs)(t.td,{style:{textAlign:"right"},children:[(0,r.jsx)(t.code,{children:"1.0"})," second"]}),(0,r.jsx)(t.td,{children:"Initial delay used by exponential backoff."})]}),(0,r.jsxs)(t.tr,{children:[(0,r.jsx)(t.td,{children:(0,r.jsx)(t.code,{children:"LLM_POOL_RATE_LIMIT_RETRY_MAX_DELAY"})}),(0,r.jsxs)(t.td,{style:{textAlign:"right"},children:[(0,r.jsx)(t.code,{children:"30.0"})," seconds"]}),(0,r.jsx)(t.td,{children:"Maximum delay between rate-limit retries."})]})]})]}),"\n",(0,r.jsx)(t.p,{children:"After a rate-limit response, the pool lowers the effective concurrency for the\naffected model and retries the request. Successful c
1alls gradually restore\nthe model's concurrency up to the configured pool limit."}),"\n",(0,r.jsxs)(t.p,{children:["The pool normalizes ",(0,r.jsx)(t.code,{children:"max_pending"})," to at least the configured pool size, and\nnormalizes the retry maximum delay to at least the retry base delay."]}),"\n",(0,r.jsx)(t.p,{children:"When all retries are exhausted, the backend records the detailed failure in\nits service log and the frontend task log receives only the stage, context, and\nerror type. Provider response bodies are not forwarded to the frontend."}),"\n",(0,r.jsx)(t.h2,{id:"example",children:"Example"}),"\n",(0,r.jsx)(t.p,{children:"The following configuration lowers concurrency for providers with a smaller\nrequest limit and shortens the Wiki MAP timeout:"}),"\n",(0,r.jsx)(t.pre,{children:(0,r.jsx)(t.code,{className:"language-dotenv",children:"WIKI_MAP_LLM_POOL_SIZE=8\nWIKI_MAP_MAX_PENDING=10\nDOC_STRUCTURE_LLM_POOL_SIZE=8\nLLM_POOL_RATE_LIMIT_RETRIES=5\nWIKI_MAP_TIMEOUT=300\n"})}),"\n",(0,r.jsxs)(t.p,{children:["Restart the RAGFlow backend after changing ",(0,r.jsx)(t.code,{children:"docker/.env"}),". The settings are\nread when the Python modules are loaded and are not changed for already-running\ntasks."]})]})}function h(e={}){const{wrapper:t}={...(0,d.R)(),...e.components};return t?(0,r.jsx)(t,{...e,children:(0,r.jsx)(a,{...e})}):a(e)}},28453(e,t,n){n.d(t,{R:()=>l,x:()=>s});var i=n(96540);const r={},d=i.createContext(r);function l(e){const t=i.useContext(d);return i.useMemo(function(){return"function"==typeof e?e(t):{...t,...e}},[t,e])}function s(e){let t;return t=e.disableParentContext?"function"==typeof e.components?e.components(r):e.components||r:l(e.components),i.createElement(d.Provider,{value:t},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.