PageSourceSearch

https://onnxruntime.ai/_app/immutable/nodes/14.a1e480c5.js

js onnxruntime.ai collected 2026-09-24 08:28:15 UTC 23,601 bytes, 97 lines download raw bytes

1import{s as we,J as Nt,K as de,f as s,a as l,H as he,g as o,D as p,c as i,h as ke,u as fe,d as n,j as K,i as a,C as Ce}from"../chunks/scheduler.e2b5ccef.js";import{S as ye,i as be,b as Le,d as ze,m as Te,a as Oe,t as He,e as Me}from"../chunks/index.67416819.js";import{g as Pe,a as ve}from"../chunks/spread.8a54911c.js";import{P as Ie}from"../chunks/post.dcc11f0c.js";const qe=""+new URL("../assets/olive-flow.8975cbb3.png",import.meta.url).href,Ne=""+new URL("../assets/olive-commands.cb40f0f5.png",import.meta.url).href;function Ae(W){let r,k="👋 Introduction",h,m,u='At <a href="https://opensource.microsoft.com/blog/2023/06/26/olive-a-user-friendly-toolchain-for-hardware-aware-model-optimization/" rel="nofollow">Build 2023 Microsoft announced Olive (<strong>O</strong>NNX <strong>Live</strong>)</a>: an advanced model optimization toolkit designed to streamline the process of optimizing AI models for deployment with the ONNX runtime. As articulated in the following diagram, Olive can take models from frameworks like PyTorch or Hugging Face and output optimized ONNX models tailored for specific deployment targets.',c,d,At=`<img src="${qe}" alt="Olive workflow."/> <i>High-Level Olive Workflow. These hardware targets can include various AI accelerators (GPU, CPU) provided by major hardware vendors such as Qualcomm, AMD, Nvidia, and Intel</i>`,J,V,Y,_,Rt="Olive operates through a structured workflow consisting of a series of model optimization tasks known as <em>passes</em>. These passes can include model compression, graph capture, quantization, and graph optimization. Each pass has adjustable parameters that can be tuned to achieve optimal metrics like accuracy and latency, which are assessed by respective evaluators. The tool leverages a search strategy, employing algorithms to auto-tune either individual passes or sets of passes collectively, ensuring the best possible performance for the deployment targets.",Z,g,Et="Whilst the workflow paradigm used in Olive is very flexible, the learning curve can be challenging for AI Developers new to model optimization processes. To make model optimization more approachable, we have curated a set of Olive workflows for common scenarios and exposed them as a simple command in a <strong>new easy-to-use CLI for Olive</strong>:",tt,f,Xt=`<img src="${Ne}" alt="Olive Commands."/> <i>Mapping of new Olive CLI commands to the associated Olive workflow that is executed.</i>`,et,nt,at,x,Qt="In this blog, we’ll show you how to prepare models for the ONNX Runtime using the Olive CLI.",st,w,Ft="🚀 Getting started with the Olive CLI",ot,C,Bt="First, install Olive using pip:",lt,y,it,ge='<code class="language-bash">pip <span class="token function">install</span> olive-ai<span class="token punctuation">[</span>cpu,finetune<span class="token punctuation">]</span></code>',pt,b,Gt="🪄 Automatic optimizer",rt,L,Ut="Once you have installed Olive, try the automatic optimizer (<code>olive auto-opt</code>). In a single command, Olive will:",ut,v,Dt="<li>Download the model from Hugging Face</li> <li>Capture the model structure into an ONNX graph and convert the weights into ONNX format.</li> <li>Optimize the ONNX graph (for example, fusion)</li> <li>Quantize the model weights into int4</li>",ct,z,jt="The command to run automatic optimizer for the Llama-3.2-1B-Instruct model on CPU devices is:",mt,T,St=`<code>olive auto-opt \\
2    --model_name_or_path meta-llama/Llama-3.2-1B-Instruct \\
3    --trust_remote_code \\ 
4    --output_path optimized-model \\
5    --device cpu \\
6    --provider CPUExecutionProvider \\
7    --precision int4 \\
8    --use_model_builder True \\
9    --log_level 1
10</code>`,dt,O,$t="<p><strong>Tip:</strong> If want to target:</p> <ul><li>CUDA GPU, then update <code>--device</code> to <code>gpu</code> and <code>--provider</code> to <code>CUDAExecutionProvider</code>.</li> <li>Windows DirectML, then update <code>--device</code> to <code>gpu</code> and <code>--provider</code> to <code>DmlExecutionProvider</code>.</li></ul> <p>Olive will apply the optimizations specific to the device and provider.</p>",ht,H,Wt='With the <code>auto-opt</code> command, you can change the input model to one that is available on Hugging Face - for example, <a href="https://huggingface.co/HuggingFaceTB/SmolLM-360M-Instruct" rel="nofollow">HuggingFaceTB/SmolLM-360M-Instruct</a> - or a model that resides on local disk. It should be noted that the <code>--trust_remote_code</code> argument in <code>olive auto-opt</code> is only required for custom models in Hugging Face that are required to run code on your machine - for more details, read the <a href="https://huggingface.co/docs/transformers/model_doc/auto#transformers.AutoConfig.from_pretrained.trust_remote_code" rel="nofollow">Hugging Face documentation on <code>trust_remote_code</code></a>. Olive, will go through the same process of automatically convert
10ing (to ONNX), optimizing the graph and quantizing the weights.',kt,M,Kt="🧪 Experimenting with different quantization algorithms",ft,P,Jt='The Olive CLI allows you to experiment with many different quantization algorithms - such as AWQ, GPTQ, and QuaRot - and different implementations of those algorithms. For example, to Quantize Llama-3.2-1B-Instruct using <a href="https://arxiv.org/abs/2306.00978" rel="nofollow">Activation Aware Quantization (AWQ)</a>:',vt,I,Vt="<p><strong>Note:</strong> Your computer will need a CUDA GPU device and associated drivers installed to run AWQ, GPTQ and QuaRot quantization. Also, you should install the AutoAWQ package using:</p> <p><code>pip install autoawq</code></p>",_t,q,Yt=`<code>olive quantize \\
11    --model_name_or_path meta-llama/Llama-3.2-1B-Instruct \\
12    --algorithm awq \\
13    --output_path quantized-model \\
14    --log_level 1
15</code>`,gt,N,Zt="The quantize command will output a PyTorch model when using AWQ method, which you can convert to ONNX if you intend to use the model on the ONNX Runtime using:",xt,A,te=`<code>olive capture-onnx-graph \\
16    --model_name_or_path quantized-model/model \\
17    --use_ort_genai True \\
18    --log_level 1 \\
19</code>`,wt,R,ee="🎚️ Finetuning",Ct,E,ne="The Olive CLI also provides the tools to fine tune an AI Model on our own data for specific tasks using either LoRA or QLoRA. The following example will fine-tune Llama-3.2-1B-Instruct for phrase classification (given a phrase in English it will output a category for the phrase from joy/sad/fear/surprised).",yt,X,ae=`<code>olive finetune \\
20    --model_name_or_path meta-llama/Llama-3.2-1B-Instruct \\
21    --trust_remote_code \\
22    --output_path models/llama3.2/ft \\
23    --data_name xxyyzzz/phrase_classification \\
24    --text_template &quot;&lt;|start_header_id|&gt;user&lt;|end_header_id|&gt;\\n{phrase}&lt;|eot_id|&gt;&lt;|start_header_id|&gt;assistant&lt;|end_header_id|&gt;\\n{tone}&quot; \\
25    --method qlora \\
26    --max_steps 30 \\
27    --log_level 1 \\
28</code>`,bt,Q,se="The finetune command will output a Hugging Face PEFT adapter, which you can prepare for the ONNX runtime using:",Lt,F,oe=`<code># Step 1 - capture the ONNX graph of the base model and adapter
29olive capture-onnx-graph \\
30    --model_name_or_path models/llama3.2/ft/model \\
31    --adapter_path models/llama3.2/ft/adapter \\
32    --use_ort_genai \\
33    --output_path models/llama3.2/onnx \\
34    --log_level 1
35
36# Step 2 - Extract adapter weights from ONNX model and store in separate file for ORT
37 olive generate-adapter \\
38    --model_name_or_path models/llama3.2/onnx \\
39    --output_path adapter-onnx \\
40    --log_level 1       
41</code>`,zt,B,le="🤝 Inference your optimized AI models using the Generate API for ONNX Runtime",Tt,G,ie="The following Python code creates a simple console-based chat interface that inferences your optimized model with the Generate API for ONNX runtime.",Ot,U,pe='<p><strong>Tip:</strong> Other language bindings - such as C#, C/C++, Java - with more coming soon. For an up-to-date list, visit the <a href="https://github.com/microsoft/onnxruntime-genai" rel="nofollow">Generate API for ONNX Runtime Github page</a></p>',Ht,D,Mt,xe=`<code class="language-python"><span class="token keyword">import</span> onnxruntime_genai <span class="token keyword">as</span> og
42<span class="token keyword">import</span> numpy <span class="token keyword">as</span> np
43<span class="token keyword">import</span> os
44
45model_folder <span class="token operator">=</span> <span class="token string">"optimized-model/model"</span>
46
47<span class="token comment"># Load the base model and tokenizer</span>
48model <span class="token operator">=</span> og<span class="token punctuation">.</span>Model<span class="token punctuation">(</span>model_folder<span class="token punctuation">)</span>
49tokenizer <span class="token operator">=</span> og<span class="token punctuation">.</span>Tokenizer<span class="token punctuation">(</span>model<span class="token punctuation">)</span>
50tokenizer_stream <span class="token operator">=</span> tokenizer<span class="token punctuation">.</span>create_stream<span class="token punctuation">(</span><span class="token punctuation">)</span>
51
52<span class="token comment"># Set the max length to something sensible by default,</span>
53<span class="token comment"># since otherwise it will be set to the entire context length</span>
54search_options <span class="token operator">=</span> <span class="token punctuation">&#123;</span><span class="token punctuation">&#125;</span>
55search_options<span class="token punctuation">[</span><span class="token string">'max_length'</span><span class="token punctuation">]</span> <span class="token operator">=</span> <span class="token number">200</span>
56search_options<span class="token punctuation">[</span><span class="token string">'past_present_share_buffer'</span><span class="token punctuation">]</span> <span class="token operator">=</span> <span class="token boolean">False</span>
57
58chat_template <span class="token operator">=</span> <span class="token triple-quoted-string string">"""&lt;|begin_of_text|>&lt;|start_header_id|>system&lt;|end_header_id|>
59
60You are a helpful assistant&lt;|eot_id|>&lt;|start_header_id|>user&lt;|end_header_id|>
61
62&#123;input&#125;&lt;|eot_id|>&lt;|start_header_id|>assistant&lt;|end_header_id|>
63"""</span> 
64
65text <span class="token operator">=</span> <span class="token builtin">input</span><span class="token punctuation">(</span><span class="token string">"Input: "</span><span class="token punctuation">)</span>
66
67<span class="token comment"># Keep asking for input phrases</span>
68<span class="token keyword">while</span>
68 text <span class="token operator">!=</span> <span class="token string">"exit"</span><span class="token punctuation">:</span>
69    <span class="token keyword">if</span> <span class="token keyword">not</span> text<span class="token punctuation">:</span>
70        <span class="token keyword">print</span><span class="token punctuation">(</span><span class="token string">"Error, input cannot be empty"</span><span class="token punctuation">)</span>
71        exit
72
73    <span class="token comment"># generate prompt (prompt template + input)</span>
74    prompt <span class="token operator">=</span> <span class="token string-interpolation"><span class="token string">f'</span><span class="token interpolation"><span class="token punctuation">&#123;</span>chat_template<span class="token punctuation">.</span><span class="token builtin">format</span><span class="token punctuation">(</span><span class="token builtin">input</span><span class="token operator">=</span>text<span class="token punctuation">)</span><span class="token punctuation">&#125;</span></span><span class="token string">'</span></span>
75
76    <span class="token comment"># encode the prompt using the tokenizer</span>
77    input_tokens <span class="token operator">=</span> tokenizer<span class="token punctuation">.</span>encode<span class="token punctuation">(</span>prompt<span class="token punctuation">)</span>
78
79    params <span class="token operator">=</span> og<span class="token punctuation">.</span>GeneratorParams<span class="token punctuation">(</span>model<span class="token punctuation">)</span>
80    params<span class="token punctuation">.</span>set_search_options<span class="token punctuation">(</span><span class="token operator">**</span>search_options<span class="token punctuation">)</span>
81    params<span class="token punctuation">.</span>input_ids <span class="token operator">=</span> input_tokens
82    generator <span class="token operator">=</span> og<span class="token punctuation">.</span>Generator<span class="token punctuation">(</span>model<span class="token punctuation">,</span> params<span class="token punctuation">)</span>
83
84    <span class="token keyword">print</span><span class="token punctuation">(</span><span class="token string">"Output: "</span><span class="token punctuation">,</span> end<span class="token operator">=</span><span class="token string">''</span><span class="token punctuation">,</span> flush<span class="token operator">=</span><span class="token boolean">True</span><span class="token punctuation">)</span>
85    <span class="token comment"># stream the output</span>
86    <span class="token keyword">try</span><span class="token punctuation">:</span>
87        <span class="token keyword">while</span> <span class="token keyword">not</span> generator<span class="token punctuation">.</span>is_done<span class="token punctuation">(</span><span class="token punctuation">)</span><span class="token punctuation">:</span>
88            generator<span class="token punctuation">.</span>compute_logits<span class="token punctuation">(</span><span class="token punctuation">)</span>
89            generator<span class="token punctuation">.</span>generate_next_token<span class="token punctuation">(</span><span class="token punctuation">)</span>
90
91            new_token <span class="token operator">=</span> generator<span class="token punctuation">.</span>get_next_tokens<span class="token punctuation">(</span><span class="token punctuation">)</span><span class="token punctuation">[</span><span class="token number">0</span><span class="token punctuation">]</span>
92            <span class="token keyword">print</span><span class="token punctuation">(</span>tokenizer_stream<span class="token punctuation">.</span>decode<span class="token punctuation">(</span>new_token<span class="token punctuation">)</span><span class="token punctuation">,</span> end<span class="token operator">=</span><span class="token string">''</span><span class="token punctuation">,</span> flush<span class="token operator">=</span><span class="token boolean">True</span><span class="token punctuation">)</span>
93    <span class="token keyword">except</span> KeyboardInterrupt<span class="token punctuation">:</span>
94        <span class="token keyword">print</span><span class="token punctuation">(</span><span class="token string">"  --control+c pressed, aborting generation--"</span><span class="token punctuation">)</span>
95
96    <span class="token keyword">print</span><span class="token punctuation">(</span><span class="token punctuation">)</span>
97    text <span class="token operator">=</span> <span class="token builtin">input</span><span class="token punctuation">(</span><span class="token string">"Input: "</span><span class="token punctuation">)</span></code>`,Pt,j,re="Conclusion",It,S,ue="In this blog we demonstrated how you can compose models for the ONNX Rutime using the new Olive CLI, and then inference those models using the Generate API for ONNX Runtime. The Olive CLI commands execute a curated Olive workflow for you, meaning you continue to get all the following benefits:",qt,$,ce='<li><strong>Reduce frustration and time</strong> of trial-and-error manual experimentation with different techniques for graph optimization, compression and quantization. Define your quality and performance constraints and let Olive automatically find the best model for you.</li> <li><strong>40+ built-in model optimization components</strong> covering cutting edge techniques in quantization, compression, graph optimization and finetuning.</li> <li>Supports creating models so they can be served using the <strong>Multi LoRA paradigm</strong>.</li> <li><strong>Hugging Face</strong> and <strong>Azure AI</strong> Integration.</li> <li>Built-in <strong>caching</strong> mechanism to save costs and <strong>enhance team collaboration</strong>. As we shared in an earlier blog post, Olive also supports a <a href="../blogs/olive-shared-cache">shared cache</a>.</li>';return{c(){r=s("h2"),r.textContent=k,h=l(),m=s("p"),m.innerHTML=u,c=l(),d=s("div"),d.innerHTML=At,J=l(),V=s("br"),Y=l(),_=s("p"),_.innerHTML=Rt,Z=l(),g=s("p"),g.innerHTML=Et,tt=l(),f=s("div"),f.innerHTML=Xt,et=l(),nt=s("br"),at=l(),x=s("p"),x.textContent=Qt,st=l(),w=s("h2"),w.textContent=Ft,ot=l(),C=s("p"),C.textContent=Bt,lt=l(),y=s("pre"),it=new he(!1),pt=l(),b=s("h3"),b.textContent=Gt,rt=l(),L=s("p"),L.innerHTML=Ut,ut=l(),v=s("ol"),v.innerHTML=Dt,ct=l(),z=s("p"),z.textContent=jt,mt=l(),T=s("pre"),T.innerHTML=St,dt=l(),O=s("blockquote"),O.innerHTML=$t,ht=l(),H=s("p"),H.innerHTML=Wt,kt=l(),M=s("h3"),M.textContent=Kt,ft=l(),P=s("p"),P.innerHTML=Jt,vt=l(),I=s("blockquote"),I.innerHTML=Vt,_t=l(),q=s("pre"),q.innerHTML=Yt,gt=l(),N=s("p"),N.textContent=Zt,xt=l(),A=s("pre"),A.innerHTML=te,wt=l(),R=s("h3"),R.textContent=ee,Ct=l(),E=s("p"),E.textContent=ne,yt=l(),X=s("pre"),X.innerHTML=ae,bt=l(),Q=s("p"),Q.textContent=se,Lt=l(),F=s("pre"),F.innerHTML=oe,zt=l(),B=s("h3"),B.textContent=le,Tt=l(),G=s("p"),G.textContent=ie,Ot=l(),U=s("blockquote"),U.innerHTML=pe,Ht=l(),D=s("pre"),Mt=new he(!1),Pt=l(),j=s("h2"),j.textContent=re,It=l(),S=s("p"),S.textContent=ue,qt=l(),$=s("ul"),$.innerHTML=ce,this.h()},l(t){r=o(t,"H2",{"data-svelte-h":!0}),p(r)!=="svelte-ekz7jq"&&(r.textContent=k),h=i(t),m=o(t,"P",{"data-svelte-h":!0}),p(m)!=="svelte-8e3zjs"&&(m.innerHTML=u),c=i(t),d=o(t,"DIV",{class:!0,"data-svelte-h":!0}),p(d)!=="svelte-1k3zve9"&&(d.innerHTML=At),J=i(t),V=o(t,"BR",{}),Y=i(t),_=o(t,"P",{"data-svelte-h":!0}),p(_)!=="svelte-4zqqq8"&&(_.innerHTML=Rt),Z=i(t),g=o(t,"P",{"data-svelte-h":!0}),p(g)!=="svelte-cufqsx"&&(g.innerHTML=Et),tt=i(t),f=o(t,"DIV",{class:!0,"data-svelte-h":!0}),p(f)!=="svelte-8p5ns8"&&(f.innerHTML=Xt),et=i(t),nt=o(t,"BR",{}),at=i(t),x=o(t,"P",{"data-svelte-h":!0}),p(x)!=="svelte-1o05u91"&&(x.textContent=Qt),st=i(t),w=o(t,"H2",{"data-svelte-h":!0}),p(w)!=="svelte-171i9mo"&&(w.textContent=Ft),ot=i(t),C=o(t,"P",{"data-svelte-h":!0}),p(C)!=="svelte-1xpybdf"&&(C.textContent=Bt),lt=i(t),y=o(t,"PRE",{class:!0});var e=ke(y);it=fe(e,!1),e.forEach(n),pt=i(t),b=o(t,"H3",{"data-svelte-h":!0}),p(b)!=="svelte-kw5dsa"&&(b.textContent=Gt),rt=i(t),L=o(t,"P",{"data-svelte-h":!0}),p(L)!=="svelte-1ukduv1"&&(L.innerHTML=Ut),ut=i(t),v=o(t,"OL",{class:!0,"data-svelte-h":!0}),p(v)!=="svelte-r10wyd"&&(v.innerHTML=Dt),ct=i(t),z=o(t,"P",{"data-svelte-h":!0}),p(z)!=="svelte-s7rxkb"&&(z.textContent=jt),mt=i(t),T=o(t,"PRE",{"data-svelte-h":!0}),p(T)!=="svelte-1cdnws9"&&(T.innerHTML=St),dt=i(t),O=o(t,"BLOCKQUOTE",{"data-svelte-h":!0}),p(O)!=="svelte-1vkmsoi"&&(O.innerHTML=$t),ht=i(t),H=o(t,"P",{"data-svelte-h":!0}),p(H)!=="svelte-551mxy"&&(H.innerHTML=Wt),kt=i(t),M=o(t,"H3",{"data-svelte-h":!0}),p(M)!=="svelte-liftgv"&&(M.textContent=Kt),ft=i(t),P=o(t,"P",{"data-svelte-h":!0}),p(P)!=="svelte-1wcu9u9"&&(P.innerHTML=Jt),vt=i(t),I=o(t,"BLOCKQUOTE",{"data-svelte-h":!0}),p(I)!=="svelte-1q6p162"&&(I.innerHTML=Vt),_t=i(t),q=o(t,"PRE",{"data-svelte-h":!0}),p(q)!=="svelte-jh9zp2"&&(q.innerHTML=Yt),gt=i(t),N=o(t,"P",{"data-svelte-h":!0}),p(N)!=="svelte-10svbhl"&&(N.textContent=Zt),xt=i(t),A=o(t,"PRE",{"data-svelte-h":!0}),p(A)!=="svelte-1fa5mi9"&&(A.innerHTML=te),wt=i(t),R=o(t,"H3",{"data-svelte-h":!0}),p(R)!=="svelte-19of08"&&(R.textContent=ee),Ct=i(t),E=o(t,"P",{"data-svelte-h":!0}),p(E)!=="svelte-qq23rf"&&(E.textContent=ne),yt=i(t),X=o(t,"PRE",{"data-svelte-h":!0}),p(X)!=="svelte-17mwtuo"&&(X.innerHTML=ae),bt=i(t),Q=o(t,"P",{"data-svelte-h":!0}),p(Q)!=="svelte-18cwby6"&&(Q.textContent=se),Lt=i(t),F=o(t,"PRE",{"data-svelte-h":!0}),p(F)!=="svelte-d8rpn3"&&(F.innerHTML=oe),zt=i(t),B=o(t,"H3",{"data-svelte-h":!0}
97),p(B)!=="svelte-1kbi4eu"&&(B.textContent=le),Tt=i(t),G=o(t,"P",{"data-svelte-h":!0}),p(G)!=="svelte-wgi01f"&&(G.textContent=ie),Ot=i(t),U=o(t,"BLOCKQUOTE",{"data-svelte-h":!0}),p(U)!=="svelte-1xyux03"&&(U.innerHTML=pe),Ht=i(t),D=o(t,"PRE",{class:!0});var me=ke(D);Mt=fe(me,!1),me.forEach(n),Pt=i(t),j=o(t,"H2",{"data-svelte-h":!0}),p(j)!=="svelte-grw4hp"&&(j.textContent=re),It=i(t),S=o(t,"P",{"data-svelte-h":!0}),p(S)!=="svelte-ruzi0g"&&(S.textContent=ue),qt=i(t),$=o(t,"UL",{"data-svelte-h":!0}),p($)!=="svelte-1s0jmjb"&&($.innerHTML=ce),this.h()},h(){K(d,"class","m-auto w55"),K(f,"class","m-auto w55"),it.a=null,K(y,"class","language-bash"),K(v,"class","svelte-d7szzl"),Mt.a=null,K(D,"class","language-python")},m(t,e){a(t,r,e),a(t,h,e),a(t,m,e),a(t,c,e),a(t,d,e),a(t,J,e),a(t,V,e),a(t,Y,e),a(t,_,e),a(t,Z,e),a(t,g,e),a(t,tt,e),a(t,f,e),a(t,et,e),a(t,nt,e),a(t,at,e),a(t,x,e),a(t,st,e),a(t,w,e),a(t,ot,e),a(t,C,e),a(t,lt,e),a(t,y,e),it.m(ge,y),a(t,pt,e),a(t,b,e),a(t,rt,e),a(t,L,e),a(t,ut,e),a(t,v,e),a(t,ct,e),a(t,z,e),a(t,mt,e),a(t,T,e),a(t,dt,e),a(t,O,e),a(t,ht,e),a(t,H,e),a(t,kt,e),a(t,M,e),a(t,ft,e),a(t,P,e),a(t,vt,e),a(t,I,e),a(t,_t,e),a(t,q,e),a(t,gt,e),a(t,N,e),a(t,xt,e),a(t,A,e),a(t,wt,e),a(t,R,e),a(t,Ct,e),a(t,E,e),a(t,yt,e),a(t,X,e),a(t,bt,e),a(t,Q,e),a(t,Lt,e),a(t,F,e),a(t,zt,e),a(t,B,e),a(t,Tt,e),a(t,G,e),a(t,Ot,e),a(t,U,e),a(t,Ht,e),a(t,D,e),Mt.m(xe,D),a(t,Pt,e),a(t,j,e),a(t,It,e),a(t,S,e),a(t,qt,e),a(t,$,e)},p:Ce,d(t){t&&(n(r),n(h),n(m),n(c),n(d),n(J),n(V),n(Y),n(_),n(Z),n(g),n(tt),n(f),n(et),n(nt),n(at),n(x),n(st),n(w),n(ot),n(C),n(lt),n(y),n(pt),n(b),n(rt),n(L),n(ut),n(v),n(ct),n(z),n(mt),n(T),n(dt),n(O),n(ht),n(H),n(kt),n(M),n(ft),n(P),n(vt),n(I),n(_t),n(q),n(gt),n(N),n(xt),n(A),n(wt),n(R),n(Ct),n(E),n(yt),n(X),n(bt),n(Q),n(Lt),n(F),n(zt),n(B),n(Tt),n(G),n(Ot),n(U),n(Ht),n(D),n(Pt),n(j),n(It),n(S),n(qt),n($))}}}function Re(W){let r,k;const h=[W[0],_e];let m={$$slots:{default:[Ae]},$$scope:{ctx:W}};for(let u=0;u<h.length;u+=1)m=Nt(m,h[u]);return r=new Ie({props:m}),{c(){Le(r.$$.fragment)},l(u){ze(r.$$.fragment,u)},m(u,c){Te(r,u,c),k=!0},p(u,[c]){const d=c&1?Pe(h,[c&1&&ve(u[0]),c&0&&ve(_e)]):{};c&2&&(d.$$scope={dirty:c,ctx:u}),r.$set(d)},i(u){k||(Oe(r.$$.fragment,u),k=!0)},o(u){He(r.$$.fragment,u),k=!1},d(u){Me(r,u)}}}const _e={title:"Democratizing AI Model optimization with the new Olive CLI",date:"11th November, 2024",description:"Learn how to use the new Olive CLI to easily optimize AI Models for on-device inference",keywords:"onnx, onnx runtime, olive, machine learning, ml, ai, quantization, on-device, real-time, mobile apps, recommendation systems, privacy, performance, cost-efficient, phi-3, small, medium, models, phi-3s-onnx, phi-3m-onnx, phi-3l-onnx, phi-3xl-onnx, phi-3xxl-onnx, phi-3s-onnx-optimized, phi-3m-onnx-optimized, phi-3l-onnx-optimized, phi-3xl-onnx-optimized, phi-3xxl-onnx-optimized, llama-3.2",authors:["Jambay Kinley","Hitesh Shah","Xiaoyu Zhang","Devang Patel","Sam Kemp"],authorsLink:["https://www.linkedin.com/in/jambayk/","","https://www.linkedin.com/in/xiaoyu-zhang/","https://www.linkedin.com/in/devangpatel/","https://www.linkedin.com/in/samuel-kemp-a9253724/"],image:"https://iili.io/2uu6zG4.png",imageSquare:"https://iili.io/2uu6zG4.png",url:"https://onnxruntime.ai/blogs/olive-cli"};function Ee(W,r,k){return W.$$set=h=>{k(0,r=Nt(Nt({},r),de(h)))},r=de(r),[r]}class Ge extends ye{constructor(r){super(),be(this,r,Ee,Re,we,{})}}export{Ge as component};

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.