1"use strict";(globalThis.webpackChunkmellea_docs=globalThis.webpackChunkmellea_docs||[]).push([[32788],{68637(e,n,t){t.r(n),t.d(n,{assets:()=>r,contentTitle:()=>d,default:()=>h,frontMatter:()=>l,metadata:()=>s,toc:()=>a});const s=JSON.parse('{"id":"how-to/configure-model-options","title":"Configure model options","description":"Set temperature, seed, max tokens, system prompts, and other backend parameters at session level or per call.","source":"@site/docs/how-to/configure-model-options.md","sourceDirName":"how-to","slug":"/how-to/configure-model-options","permalink":"/next/how-to/configure-model-options","draft":false,"unlisted":false,"editUrl":"https://github.com/generative-computing/mellea/edit/main/docs/how-to/configure-model-options.md","tags":[],"version":"current","lastUpdatedAt":1789565696000,"frontMatter":{"title":"Configure model options","description":"Set temperature, seed, max tokens, system prompts, and other backend parameters at session level or per call."},"sidebar":"docsSidebar","previous":{"title":"Evaluate with LLM-as-a-Judge","permalink":"/next/how-to/evaluate-with-llm-as-a-judge"},"next":{"title":"Use Images and Vision Models","permalink":"/next/how-to/use-images-and-vision"}}');var o=t(74848),i=t(28453);const l={title:"Configure model options",description:"Set temperature, seed, max tokens, system prompts, and other backend parameters at session level or per call."},d=void 0,r={},a=[{value:"The ModelOption enum",id:"the-modeloption-enum",level:2},{value:"Precedence rules",id:"precedence-rules",level:2},{value:"Pushing and popping model state",id:"pushing-and-popping-model-state",level:2},{value:"Reference: all ModelOption keys",id:"reference-all-modeloption-keys",level:2},{value:"Streaming timeout",id:"streaming-timeout",level:2},{value:"Reasoning and thinking mode",id:"reasoning-and-thinking-mode",level:2},{value:"System prompts",id:"system-prompts",level:2}];function c(e){const n={a:"a",blockquote:"blockquote",code:"code",h2:"h2",li:"li",ol:"ol",p:"p",pre:"pre",strong:"strong",table:"table",tbody:"tbody",td:"td",th:"th",thead:"thead",tr:"tr",...(0,i.R)(),...e.components};return(0,o.jsxs)(o.Fragment,{children:[(0,o.jsxs)(n.p,{children:["Most LLM APIs accept parameters such as temperature, max tokens, and seed. Mellea exposes\nthese through the ",(0,o.jsx)(n.code,{children:"ModelOption"})," enum, which works uniformly across all backends, and also\nlets you pass backend-native keys directly."]}),"\n",(0,o.jsxs)(n.p,{children:[(0,o.jsx)(n.strong,{children:"Prerequisites:"})," ",(0,o.jsx)(n.code,{children:"pip install mellea"})," complete, a backend available (see\n",(0,o.jsx)(n.a,{href:"/next/getting-started/installation",children:"Installation"}),")."]}),"\n",(0,o.jsx)(n.h2,{id:"the-modeloption-enum",children:"The ModelOption enum"}),"\n",(0,o.jsxs)(n.p,{children:["Import ",(0,o.jsx)(n.code,{children:"ModelOption"})," from ",(0,o.jsx)(n.code,{children:"mellea.backends"}),". The enum provides cross-backend names\nfor the most common parameters:"]}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-python",children:'import mellea\nfrom mellea.backends import ModelOption, model_ids\nfrom mellea.backends.ollama import OllamaModelBackend\n\nm = mellea.MelleaSession(\n backend=OllamaModelBackend(\n model_id=model_ids.IBM_GRANITE_4_HYBRID_SMALL,\n model_options={ModelOption.SEED: 42},\n )\n)\n\nanswer = m.instruct(\n "What is 2x2?",\n model_options={\n ModelOption.TEMPERATURE: 0.5,\n ModelOption.MAX_NEW_TOKENS: 10,\n },\n)\nprint(str(answer))\n# Output will vary \u2014 LLM responses depend on model and temperature.\n'})}),"\n",(0,o.jsxs)(n.p,{children:["Options set on the backend apply to every call on that session. Options passed to a specific\n",(0,o.jsx)(n.code,{children:"m.*"})," call apply only to that call and take precedence over the session-level values."]}),"\n",(0,o.jsx)(n.p,{children:"You can also pass backend-native key names directly \u2014 Mellea forwards any key it does not\nrecognize to the underlying API unchanged. This means you can copy model option dicts from\nexisting codebase
1s without translation:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-python",children:'answer = m.instruct(\n "Summarize this in one sentence.",\n model_options={\n "temperature": 0.3,\n "num_predict": 50, # Ollama-native key\n },\n)\n'})}),"\n",(0,o.jsx)(n.h2,{id:"precedence-rules",children:"Precedence rules"}),"\n",(0,o.jsx)(n.p,{children:"When the same option is set in multiple places, the following rules apply:"}),"\n",(0,o.jsxs)(n.ol,{children:["\n",(0,o.jsxs)(n.li,{children:["A ",(0,o.jsx)(n.code,{children:"ModelOption"})," key always takes precedence over its backend-native equivalent."]}),"\n",(0,o.jsxs)(n.li,{children:["Options passed to a ",(0,o.jsx)(n.code,{children:"m.*"})," call override the corresponding session-level options for that\ncall only."]}),"\n"]}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-python",children:'# Backend initialised with these options\nbackend_options = {\n "seed": 1,\n ModelOption.MAX_NEW_TOKENS: 100,\n "temperature": 1.0,\n}\n\n# Options passed at call time\ncall_options = {\n "seed": 2,\n ModelOption.SEED: 3, # takes precedence over "seed": 2\n "num_predict": 50,\n}\n\n# Options actually sent to the model for this call:\n# seed = 3 (ModelOption.SEED wins)\n# max_new_tokens = 100 (from backend; not overridden)\n# temperature = 1.0 (from backend; not overridden)\n# num_predict = 50 (new key from call)\n'})}),"\n",(0,o.jsx)(n.h2,{id:"pushing-and-popping-model-state",children:"Pushing and popping model state"}),"\n",(0,o.jsx)(n.p,{children:"Sessions support temporarily overriding model options for a series of calls, then restoring\nthe original state:"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-python",children:'m = mellea.start_session()\n\nm.push_model_options({ModelOption.TEMPERATURE: 0.0, ModelOption.SEED: 99})\n\n# These calls use temperature=0.0, seed=99\nresult1 = m.instruct("List three capitals of South America.")\nresult2 = m.instruct("List three capitals of Europe.")\n\nm.pop_model_options()\n\n# Back to original session options\nresult3 = m.instruct("Write a short poem.")\n'})}),"\n",(0,o.jsx)(n.p,{children:"This is useful when you need deterministic output for a batch of calls within a larger,\nnon-deterministic session."}),"\n",(0,o.jsx)(n.h2,{id:"reference-all-modeloption-keys",children:"Reference: all ModelOption keys"}),"\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n",(0,o.jsxs)(n.table,{children:[(0,o.jsx)(n.thead,{children:(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.th,{children:"Key"}),(0,o.jsx)(n.th,{children:"Type"}),(0,o.jsx)(n.th,{children:"Default"}),(0,o.jsx)(n.th,{children:"Description"})]})}),(0,o.jsxs)(n.tbody,{children:[(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"ModelOption.TEMPERATURE"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"float"})}),(0,o.jsx)(n.td,{children:"backend default"}),(0,o.jsx)(n.td,{children:"Sampling temperature."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"ModelOption.MAX_NEW_TOKENS"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"int"})}),(0,o.jsx)(n.td,{children:"backend default \u26a0\ufe0f"}),(0,o.jsx)(n.td,{children:"Maximum tokens to generate. Backend defaults vary widely \u2014 set this explicitly in production code."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"ModelOption.SEED"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"int"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"None"})}),(0,o.jsx)(n.td,{children:"Random seed for reproducible output."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"ModelOption.SYSTEM_PROMPT"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"str"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"None"})}),(0,o.jsx)(n.td,{children:"System prompt prepended to every call on the session."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"ModelOption.STREAM"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"bool"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"False"})}),(0,o.jsx)(n.td,{children:"Enable streaming output."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"ModelOption.STREAM_TIMEOUT"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"float | None"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"120.0"})}),(0,o.jsxs)(n.td,{children:["Timeout in seconds applied to every chunk, including time-to-first-token. Only applies to streaming responses; non-streaming calls are unaffected. If no chunk arrives within this window the stream aborts with a ",(0,o.jsx)(n.code,{children:"TimeoutError"}),". Set to ",(0,o.jsx)(n.code,{children:"None"})," to disable. Increase for slow local inference."]})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"ModelOption.STOP_SEQUENCES"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"list[str]"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"None"})}),(0,o.jsx)(n.td,{children:"Strings that halt generation when produced by the model."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"ModelOption.THINKING"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"bool | str"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"None"})}),(0,o.jsxs)(n.td,{children:["Enable or configure reasoning/thinking mode. See ",(0,o.jsx)(n.a,{href:"#reasoning-and-thinking-mode",children:"Reasoning and thinking mode"})," below."]})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"ModelOption.CONTEXT_WINDOW"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"int"})}),(0,o.jsx)(n.td,{children:"backend default"}),(0,o.jsx)(n.td,{children:"Context window size override."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"ModelOption.TOOLS"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"list[MelleaTool]"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"None"})}
1),(0,o.jsx)(n.td,{children:"Tools exposed to the model for tool calling."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"ModelOption.TOOL_CHOICE"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"str"})}),(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:'"auto"'})}),(0,o.jsxs)(n.td,{children:["Tool selection strategy (",(0,o.jsx)(n.code,{children:'"none"'}),", ",(0,o.jsx)(n.code,{children:'"auto"'}),", or a specific tool name)."]})]})]})]}),"\n",(0,o.jsx)(n.p,{children:"Keys marked with a backend default are forwarded to the underlying API unchanged; the\nvalue the model sees depends on the backend's own defaults."}),"\n",(0,o.jsxs)(n.blockquote,{children:["\n",(0,o.jsxs)(n.p,{children:[(0,o.jsx)(n.strong,{children:"Warning:"})," ",(0,o.jsx)(n.code,{children:"MAX_NEW_TOKENS"})," backend defaults vary widely and some are very low \u2014 for\nexample vLLM defaults to 16 tokens, which will silently truncate most real responses.\nAlways set ",(0,o.jsx)(n.code,{children:"ModelOption.MAX_NEW_TOKENS"})," explicitly in production code:"]}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-python",children:'m.instruct(\n "Summarise this document.",\n model_options={ModelOption.MAX_NEW_TOKENS: 1024},\n)\n'})}),"\n",(0,o.jsx)(n.p,{children:"A value of 512\u20132048 covers most chat and instruction use cases. For code generation\nor long-form output, set a higher value to match your expected output length."}),"\n"]}),"\n",(0,o.jsx)(n.h2,{id:"streaming-timeout",children:"Streaming timeout"}),"\n",(0,o.jsxs)(n.p,{children:["By default Mellea waits up to 120 seconds for each chunk, including the first\n(time-to-first-token). If the backend stops sending without closing the connection\nthe stream aborts with a ",(0,o.jsx)(n.code,{children:"TimeoutError"})," rather than hanging indefinitely. This\ntimeout only applies to streaming responses; non-streaming calls are unaffected."]}),"\n",(0,o.jsxs)(n.blockquote,{children:["\n",(0,o.jsxs)(n.p,{children:[(0,o.jsx)(n.strong,{children:"Note for slow or local backends:"})," The 120 s default covers time-to-first-token.\nLarge models on CPU, long prompts, or heavily loaded servers can take longer than\nthis before producing the first token. Use a higher value or ",(0,o.jsx)(n.code,{children:"None"})," for those\ndeployments."]}),"\n"]}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-python",children:'from mellea.backends import ModelOption\n\n# Tighter bound for a known-fast remote endpoint\nmot = await m.ainstruct(\n "Summarise this document.",\n model_options={ModelOption.STREAM: True, ModelOption.STREAM_TIMEOUT: 10},\n)\n\n# Larger value for slow local inference (e.g. large model on CPU)\nmot = await m.ainstruct(\n "Write a long analysis.",\n model_options={ModelOption.STREAM: True, ModelOption.STREAM_TIMEOUT: 300},\n)\n\n# Disable entirely \u2014 original unbounded behaviour\nmot = await m.ainstruct(\n "Write a long analysis.",\n model_options={ModelOption.STREAM: True, ModelOption.STREAM_TIMEOUT: None},\n)\n'})}),"\n",(0,o.jsx)(n.h2,{id:"reasoning-and-thinking-mode",children:"Reasoning and thinking mode"}),"\n",(0,o.jsxs)(n.p,{children:[(0,o.jsx)(n.code,{children:"ModelOption.THINKING"})," enables or configures a model's reasoning/thinking mode.\nAccepted values and their effect are backend-dependent:"]}),"\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n",(0,o.jsxs)(n.table,{children:[(0,o.jsx)(n.thead,{children:(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.th,{children:"Backend"}),(0,o.jsx)(n.th,{children:(0,o.jsx)(n.code,{children:"True"})}),(0,o.jsx)(n.th,{children:(0,o.jsx)(n.code,{children:"False"})}),(0,o.jsxs)(n.th,{children:[(0,o.jsx)(n.code,{children:'"low"'})," / ",(0,o.jsx)(n.code,{children:'"medium"'})," / ",(0,o.jsx)(n.code,{children:'"high"'})]})]})}),(0,o.jsxs)(n.tbody,{children:[(0,o.jsxs)(n.tr,{children:[(0,o.jsxs)(n.td,{children:["Native ",(0,o.jsx)(n.code,{children:"OllamaModelBackend"})]}),(0,o.jsxs)(n.td,{children:["Enables thinking (Ollama ",(0,o.jsx)(n.code,{children:"think=True"}),")"]}),(0,o.jsxs)(n.td,{children:["Disables thinking (",(0,o.jsx)(n.code,{children:"think=False"}),")"]}),(0,o.jsxs)(n.td,{children:["Passed through to Ollama's ",(0,o.jsx)(n.code,{children:"think="})," param, which handles string effort levels itself"]})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsxs)(n.td,{children:[(0,o.jsx)(n.code,{children:"OpenAIBacken
1d"})," / LiteLLM (OpenAI-compatible)"]}),(0,o.jsxs)(n.td,{children:["Enables thinking (",(0,o.jsx)(n.code,{children:'reasoning_effort="medium"'}),", plus ",(0,o.jsx)(n.code,{children:"chat_template_kwargs.enable_thinking=True"})," for vLLM-served templates)"]}),(0,o.jsxs)(n.td,{children:["Disables thinking on vLLM-served/OpenAI-compatible servers that honour ",(0,o.jsx)(n.code,{children:"chat_template_kwargs"}),". ",(0,o.jsxs)(n.strong,{children:["Real OpenAI reasoning models, and LiteLLM targets that aren't Ollama, deliberately never receive ",(0,o.jsx)(n.code,{children:'reasoning_effort="none"'})]})," (real OpenAI rejects that value) \u2014 there is no supported way to fully disable reasoning on them via ",(0,o.jsx)(n.code,{children:"ModelOption.THINKING"})]}),(0,o.jsxs)(n.td,{children:["Sent as ",(0,o.jsx)(n.code,{children:"reasoning_effort"})," verbatim \u2014 a top-level request parameter, independent of any chat template"]})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"LocalHFBackend"})}),(0,o.jsxs)(n.td,{children:["Forwards to whichever chat-template variable is declared (",(0,o.jsx)(n.code,{children:"think"}),", ",(0,o.jsx)(n.code,{children:"thinking"}),", or ",(0,o.jsx)(n.code,{children:"enable_thinking"}),")"]}),(0,o.jsxs)(n.td,{children:["Same, ",(0,o.jsx)(n.code,{children:"False"})," value"]}),(0,o.jsxs)(n.td,{children:["Forwarded verbatim as the chat template's own ",(0,o.jsx)(n.code,{children:"reasoning_effort"})," variable when the template declares one. This is a different transport than the OpenAI backend's top-level parameter \u2014 it only takes effect if the served model's template exposes that variable"]})]})]})]}),"\n",(0,o.jsxs)(n.p,{children:["For Granite 4.2 specifically: the chat template only distinguishes ",(0,o.jsx)(n.code,{children:'"low"'}),"\neffort from everything else \u2014 ",(0,o.jsx)(n.code,{children:'reasoning_effort == "low"'})," triggers genuine\nlow-effort (short) reasoning, while ",(0,o.jsx)(n.code,{children:'"medium"'}),"/",(0,o.jsx)(n.code,{children:'"high"'})," are accepted but\nbehave the same as ",(0,o.jsx)(n.code,{children:"True"}),"/omitted (full-length reasoning). This holds across\nall three backends, provided the serving runtime forwards the effort level\ninto the chat template (Ollama and vLLM do). Granite defaults to thinking\n",(0,o.jsx)(n.strong,{children:"on"})," when ",(0,o.jsx)(n.code,{children:"ModelOption.THINKING"})," is not set at all."]}),"\n",(0,o.jsxs)(n.p,{children:[(0,o.jsx)(n.code,{children:"LocalHFBackend"})," also parses Granite's ",(0,o.jsx)(n.code,{children:"<think>...</think>"})," block out of the\nresponse, so ",(0,o.jsx)(n.code,{children:"result.thinking"})," and ",(0,o.jsx)(n.code,{children:"result.value"})," are populated separately \u2014\nmatching the other backends \u2014 rather than leaving the reasoning trace\nembedded raw in ",(0,o.jsx)(n.code,{children:"result.value"}),". This split is skipped for streaming (",(0,o.jsx)(n.code,{children:"m serve"}),")\ncalls, where the reasoning trace still arrives inline in ",(0,o.jsx)(n.code,{children:"result.value"}),";\nincremental splitting for streaming is tracked separately in\n",(0,o.jsx)(n.a,{href:"https://github.com/generative-computing/mellea/issues/1604",children:"#1604"}),"."]}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-python",children:'import mellea\nfrom mellea.backends import ModelOption, model_ids\nfrom mellea.backends.ollama import OllamaModelBackend\n\nm = mellea.MelleaSession(\n backend=OllamaModelBackend(model_id=model_ids.IBM_GRANITE_4_2_3B)\n)\n\n# Full reasoning (the default for Granite 4.2)\nanswer = m.instruct("What is 17 * 24?", model_options={ModelOption.THINKING: True})\nprint(answer.thinking) # reasoning trace\nprint(answer.value) # final answer\n# Output will vary \u2014 reasoning traces are non-deterministic.\n\n# Short, low-effort reasoning \u2014 use when you need an answer within a small\n# token budget rather than an exhaustive trace\nanswer = m.instruct("What is 17 * 24?", model_options={ModelOption.THINKING: "low"})\n\n# No reasoning at all\nanswer = m.instruct("What is 17 * 24?", model_options={ModelOption.THINKING: False})\n'})}),"\n",(0,o.jsxs)(n.blockquote,{children:["\n",(0,o.jsxs)(n.p,{children:[(0,o.jsx)(n.strong,{children:"Full example:"})," ",(0,o.jsx)(n.a,{href:"https://github.com/generative-computing/mellea/blob/main/docs/examples/thinking_mode.py",children:(0,o.jsx)(n.code,{children:"docs/examples/thinking_mode.py"})})]}),"\n"]}),"\n",(0,o.jsx)(n.p,{children:"If you're serving Granite 4.2 via vLLM, make sure your vLLM install picks up\nthe model's latest reasoning-parser update (shipped in the model's Hugging\nFace files) \u2014 an older cached parser produces stale thinking behavior."}),"\n",(0,o.jsxs)(n.p,{children:["For non-Granite thinking models served through an OpenAI-compatible endpoint\n(e.g. Qwe
1n3 on vLLM), see\n",(0,o.jsxs)(n.a,{href:"/next/integrations/openai#empty-value-from-a-thinking-mode-model",children:["Empty ",(0,o.jsx)(n.code,{children:"value"})," from a thinking-mode model"]}),"\nin the OpenAI integration guide."]}),"\n",(0,o.jsx)(n.h2,{id:"system-prompts",children:"System prompts"}),"\n",(0,o.jsxs)(n.p,{children:["Set a system prompt with ",(0,o.jsx)(n.code,{children:"ModelOption.SYSTEM_PROMPT"}),". At session level it applies to all\nsubsequent calls; at call level it applies only to that call."]}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-python",children:'m = mellea.MelleaSession(\n backend=OllamaModelBackend(\n model_id=model_ids.IBM_GRANITE_4_HYBRID_MICRO,\n model_options={\n ModelOption.SYSTEM_PROMPT: "You are a concise technical assistant. Never use bullet points."\n },\n )\n)\n\nanswer = m.instruct("Explain what a context manager is in Python.")\n'})}),"\n",(0,o.jsxs)(n.p,{children:["Using ",(0,o.jsx)(n.code,{children:"ModelOption.SYSTEM_PROMPT"})," is recommended over constructing a system-role message\nmanually. Some backend APIs do not serialize system-role messages correctly and expect the\nsystem prompt as a separate parameter \u2014 ",(0,o.jsx)(n.code,{children:"ModelOption.SYSTEM_PROMPT"})," handles this correctly\nacross all backends."]})]})}function h(e={}){const{wrapper:n}={...(0,i.R)(),...e.components};return n?(0,o.jsx)(n,{...e,children:(0,o.jsx)(c,{...e})}):c(e)}},28453(e,n,t){t.d(n,{R:()=>l,x:()=>d});var s=t(96540);const o={},i=s.createContext(o);function l(e){const n=s.useContext(i);return s.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function d(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(o):e.components||o:l(e.components),s.createElement(i.Provider,{value:n},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.