1"use strict";(globalThis.webpackChunk_project_hami_website||=[]).push([[5325],{13072(e,n,i){i.r(n),i.d(n,{assets:()=>r,contentTitle:()=>c,default:()=>l,frontMatter:()=>t,metadata:()=>o,toc:()=>d});const o=JSON.parse('{"id":"developers/dynamic-mig","title":"dynamic-mig","description":"----","source":"@site/versioned_docs/version-v2.4.1/developers/dynamic-mig.md","sourceDirName":"developers","slug":"/developers/dynamic-mig","permalink":"/docs/v2.4.1/developers/dynamic-mig","draft":false,"unlisted":false,"editUrl":"https://github.com/Project-HAMi/website/edit/master/versioned_docs/version-v2.4.1/developers/dynamic-mig.md","tags":[],"version":"v2.4.1","lastUpdatedAt":1783504905000,"frontMatter":{},"sidebar":"docs","previous":{"title":"HAMi-core design","permalink":"/docs/v2.4.1/developers/hami-core-design"},"next":{"title":"HAMi WebUI Developer Guide","permalink":"/docs/v2.4.1/developers/hami-webui-development-guide"}}');var a=i(74848),s=i(28453);const t={},c=void 0,r={},d=[{value:"Dynamic MIG Implementation",id:"dynamic-mig-implementation",level:2},{value:"Special Thanks",id:"special-thanks",level:2},{value:"Introduction",id:"introduction",level:2},{value:"Targets",id:"targets",level:2},{value:"Config maps",id:"config-maps",level:3},{value:"Structure",id:"structure",level:2},{value:"Examples",id:"examples",level:2},{value:"Procedures",id:"procedures",level:2}];function m(e){const n={a:"a",code:"code",h1:"h1",h2:"h2",h3:"h3",hr:"hr",li:"li",p:"p",pre:"pre",ul:"ul",...(0,s.R)(),...e.components};return(0,a.jsxs)(a.Fragment,{children:[(0,a.jsx)(n.hr,{}),"\n",(0,a.jsx)(n.h2,{id:"dynamic-mig-implementation",children:"Dynamic MIG Implementation"}),"\n",(0,a.jsx)(n.h1,{id:"nvidia-gpu-mps-and-mig-dynamic-slice-plugin",children:"NVIDIA GPU MPS and MIG dynamic slice plugin"}),"\n",(0,a.jsx)(n.h2,{id:"special-thanks",children:"Special Thanks"}),"\n",(0,a.jsx)(n.p,{children:"This feature will not be implemented without the help of @sailorvii."}),"\n",(0,a.jsx)(n.h2,{id:"introduction",children:"Introduction"}),"\n",(0,a.jsxs)(n.p,{children:["The NVIDIA GPU build-in sharing method includes: time-slice, MPS and MIG. The context switch for time slice sharing would waste some time, MPS and MIG are preferred. The GPU MIG profile is variable, the user could acquire the MIG device in the profile definition, but current implementation only defines the dedicated profile before the user requirement. That limits the usage of MIG. The goal is an automatic slice plugin that creates slices on demand.\nFor the scheduling method, node-level binpack and spread will be supported. Referring to the binpack plugin, the scheduler considers CPU, memory, GPU memory, and other user-defined resources.\nHAMi is done by using ",(0,a.jsx)(n.a,{href:"https://github.com/Project-HAMi/HAMi-core",children:"hami-core"}),", which is a cuda-hacking library. But MIG is also widely used across the world. A unified API for dynamic-mig and hami-core is needed."]}),"\n",(0,a.jsx)(n.h2,{id:"targets",children:"Targets"}),"\n",(0,a.jsxs)(n.ul,{children:["\n",(0,a.jsx)(n.li,{children:"CPU, Mem, and GPU combined schedule"}),"\n",(0,a.jsx)(n.li,{children:"GPU dynamic slice: HAMi-core and MIG"}),"\n",(0,a.jsx)(n.li,{children:"Support node-level binpack and spread by GPU memory, CPU and Mem"}),"\n",(0,a.jsx)(n.li,{children:"A unified vGPU Pool different virtualization techniques"}),"\n",(0,a.jsx)(n.li,{children:"Tasks can choose to use MIG, use HAMi-core, or use both."}),"\n"]}),"\n",(0,a.jsx)(n.h3,{id:"config-maps",children:"Config maps"}),"\n",(0,a.jsxs)(n.ul,{children:["\n",(0,a.jsx)(n.li,{children:"hami-scheduler-device-configMap\nThis configmap defines the plugin configurations including resourceName, and MIG geometries, and node-level configurations."}),"\n"]}),"\n",(0,a.jsx)(n.pre,{children:(0,a.jsx)(n.code,{className:"language-yaml",children:'apiVersion: v1\ndata:\n device-config.yaml: |\n nvidia:\n resourceCountName: nvidia.com/gpu\n resourceMemoryName: nvidia.com/gpumem\n resourceCoreName: nvidia.com/gpucores\n knownMigGeometries:\n - models: [ "A30" ]\n allowedGeometries:\n -\n - name: 1g.6gb\n memory: 6144\n count: 4\n -\n - name: 2g.12gb\n memory: 12288\n count: 2\n -\n - name: 4g.24gb\n memory: 24576\n count: 1\n - models: [ "A100-SXM4-40GB", "A100-40GB-PCIe", "A100-PCIE-40GB", "A100-SXM4-40GB" ]\n allowedGeometries:\n -\n - name: 1g.5gb\n memory: 5120\n count: 7\n -\n - name: 2g.10gb\n memory: 10240\n count: 3\n - name: 1g.5gb\n memory: 5120\n count: 1\n -\n - name: 3g.20gb\n memory: 20480\n count: 2\n -\n - name: 7g.40gb\n memory: 40960\n count: 1\n - models: [ "A100-SXM4-80GB", "A100-80GB-PCIe", "A100-PCIE-80GB"]\n allowedGeometries:\n -\n - name: 1g.10gb\n memory: 10240\n count: 7\n -\n - name: 2g.20gb\n memory: 20480\n count: 3\n - name: 1g.10gb\n memory: 10240\n count: 1\n -\n - name: 3g.40gb\n memory: 40960\n count: 2\n -\n - name: 7g.79gb\n memory: 80896\n count: 1\n nodeconfig:\n - name: nodeA\n operatingmode: hami-core\n - name: nodeB\n operatingmode: mig\n'})}),"\n",(0,a.jsx)(n.h2,{id:"structure",children:"Structure"}),"\n",(0,a.jsx)("img",{src:"/img/docs/en/dynamic-mig/hami-dynamic-mig-structure.png",width:"600",alt:"HAMi dynamic MIG structure diagram showing vGPU Pool and Scheduler components"}),"\n",(0,a.jsx)(n.h2,{id:"examples",children:"Examples"}),"\n",(0,a.jsxs)(n.p,{children:["Dynamic MIG is compatible with HAMi tasks, as the example below:\nJust Setting ",(0,a.jsx)(n.code,{children:"nvidia.com/gpu"})," and ",(0,a.jsx)(n.code,{children:"nvidia.com/gpumem"}),"."]}),"\n",(0,a.jsx)(n.pre,{children:(0,a.jsx)(n.code,{className:"language-yaml",children:'apiVersion: v1\nkind: Pod\nmetadata:\n name: gpu-pod1\nspec:\n containers:\n - name: ubuntu-container1\n image: ubuntu:20.04\n command: ["bash", "-c", "sleep 86400"]\n resources:\n limits:\n nvidia.com/gpu: 2 # requesting 2 vGPUs\n nvidia.com/gpumem: 8000 # Each vGPU contains 8000m device memory \uff08Optional,Integer)\n'})}),"\n",(0,a.jsxs)(n.p,{children:["A task can decide only to use ",(0,a.jsx)(n.code,{children:"mig"})," or ",(0,a.jsx)(n.code,{children:"hami-core"})," by setting ",(0,a.jsx)(n.code,{children:"annotations.nvidia.com/vgpu-mode"})," to corresponding value, as the example below shows:"]}),"\n",(0,a.jsx)(n.pre,{children:(0,a.jsx)(n.code,{className:"language-yaml",children:'apiVersion: v1\nkind: Pod\nmetadata:\n name: gpu-pod1\n annotations:\n nvidia.com/vgpu-mode: "mig"\nspec:\n containers:\n - name: ubuntu-container1\n image: ubuntu:20.04\n command: ["bash", "-c", "sleep 86400"]\n resources:\n limits:\n nvidia.com/gpu: 2 # requesting 2 vGPUs\n nvidia.com/gpumem: 8000 # Each vGPU contains 8000m device memory \uff08Optional,Integer\n'})}),"\n",(0,a.jsx)(n.h2,{id:"procedures",children:"Procedures"}),"\n",(0,a.jsx)(n.p,{children:"The Procedure of a vGPU task which uses dynamic-mig is shown below:"}),"\n",(0,a.jsx)("img",{src:"/img/docs/en/dynamic-mig/hami-dynamic-mig-procedure.png",width:"800",alt:"HAMi dynamic MIG procedure flowchart showing task scheduling process"}),"\n",(0,a.jsxs)(n.p,{children:["After a task is submitted, deviceshare plugin will iterate over templates defined in configMap ",(0,a.jsx)(n.code,{children:"hami-scheduler-device"}),", and find the first available template to fit. You can always change the content of that configMap, and restart vc-scheduler to customize."]}),"\n",(0,a.jsx)(n.p,{children:"If you submit the example on an empty A100-PCIE-40GB node, then it will select a GPU and choose MIG template below:"}),"\n",(0,a.jsx)(n.pre,{children:(0,a.jsx)(n.code,{className:"language-yaml",children:" 2g.10gb : 3\n 1g.5gb : 1\n"})}
1),"\n",(0,a.jsx)(n.p,{children:"Then start the container with 2g.10gb instances * 2"})]})}function l(e={}){const{wrapper:n}={...(0,s.R)(),...e.components};return n?(0,a.jsx)(n,{...e,children:(0,a.jsx)(m,{...e})}):m(e)}},28453(e,n,i){i.d(n,{R:()=>t,x:()=>c});var o=i(96540);const a={},s=o.createContext(a);function t(e){const n=o.useContext(s);return o.useMemo((function(){return"function"==typeof e?e(n):{...n,...e}}),[n,e])}function c(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(a):e.components||a:t(e.components),o.createElement(s.Provider,{value:n},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.