PageSourceSearch

https://volcano.sh/assets/js/9e9d85c1.8d0703ac.js

js volcano.sh collected 2026-10-02 06:13:40 UTC 5,893 bytes, 1 lines download raw bytes

1"use strict";(globalThis.webpackChunkdocs_site=globalThis.webpackChunkdocs_site||[]).push([[5611],{13924(n,e,o){o.r(e),o.d(e,{assets:()=>l,contentTitle:()=>c,default:()=>p,frontMatter:()=>i,metadata:()=>s,toc:()=>a});const s=JSON.parse('{"id":"Ecosystem/TensorFlowOnVolcano","title":"TensorFlow on Volcano","description":"TensorFlow introduction","source":"@site/versioned_docs/version-v1.12.0/Ecosystem/TensorFlowOnVolcano.md","sourceDirName":"Ecosystem","slug":"/Ecosystem/TensorFlowOnVolcano","permalink":"/docs/v1.12.0/Ecosystem/TensorFlowOnVolcano","draft":false,"unlisted":false,"editUrl":"https://github.com/volcano-sh/website/tree/master/versioned_docs/version-v1.12.0/Ecosystem/TensorFlowOnVolcano.md","tags":[],"version":"v1.12.0","lastUpdatedAt":1779789566000,"sidebarPosition":7,"frontMatter":{"title":"TensorFlow on Volcano","sidebar_position":7},"sidebar":"docsSidebar","previous":{"title":"Spark on Volcano","permalink":"/docs/v1.12.0/Ecosystem/SparkOnVolcano"},"next":{"title":"Ray on Volcano","permalink":"/docs/v1.12.0/Ecosystem/RayOnVolcano"}}');var t=o(74848),r=o(28453);const i={title:"TensorFlow on Volcano",sidebar_position:7},c=void 0,l={},a=[{value:"TensorFlow introduction",id:"tensorflow-introduction",level:3},{value:"TensorFlow on Volcano",id:"tensorflow-on-volcano",level:3}];function d(n){const e={code:"code",h3:"h3",img:"img",li:"li",p:"p",pre:"pre",ul:"ul",...(0,r.R)(),...n.components};return(0,t.jsxs)(t.Fragment,{children:[(0,t.jsx)(e.h3,{id:"tensorflow-introduction",children:"TensorFlow introduction"}),"\n",(0,t.jsx)(e.p,{children:"TensorFlow is a symbolic mathematical system based on data flow programming, which is widely used in programming and realization of various machine learning algorithms. Its predecessor is DistBelief, a neural network algorithm library of Google."}),"\n",(0,t.jsx)(e.h3,{id:"tensorflow-on-volcano",children:"TensorFlow on Volcano"}),"\n",(0,t.jsx)(e.p,{children:"PS-worker model: Parameter Server performs model-related services, Work Server trains related services, inference calculation, gradient calculation, etc[1]."}),"\n",(0,t.jsx)(e.p,{children:(0,t.jsx)(e.img,{alt:"ps-worker",src:o(84927).A+"",width:"1420",height:"504"})}),"\n",(0,t.jsx)(e.p,{children:"TensorFlow on Kubernetes has many problems:"}),"\n",(0,t.jsxs)(e.ul,{children:["\n",(0,t.jsx)(e.li,{children:"Resource isolation."}),"\n",(0,t.jsx)(e.li,{children:"Lack of GPU scheduling, Gang Schuler."}),"\n",(0,t.jsx)(e.li,{children:"Process Legacy."}),"\n",(0,t.jsx)(e.li,{children:"Training log is not convenient to save."}),"\n"]}),"\n",(0,t.jsxs)(e.p,{children:["Create ",(0,t.jsx)(e.code,{children:"tftest.yaml"}),"."]}),"\n",(0,t.jsx)(e.pre,{children:(0,t.jsx)(e.code,{children:'apiVersion: batch.volcano.sh/v1alpha1\nkind: Job\nmetadata:\n  name: tensorflow-dist-mnist\nspec:\n  minAvailable: 3\n  schedulerName: volcano\n  plugins:\n    env: []\n    svc: []\n  policies:\n    - event: PodEvicted\n      action: RestartJob\n  queue: default\n  tasks:\n    - replicas: 1\n      name: ps\n      template:\n        spec:\n          containers:\n            - command:\n                - sh\n                - -c\n                - |\n                  PS_HOST=`cat /etc/volcano/ps.host | sed \'s/$/&:2222/g\' | sed \'s/^/"/;s/$/"/\' | tr "\\n" ","`;\n                  WORKER_HOST=`cat /etc/volcano/worker.host | sed \'s/$/&:2222/g\' | sed \'s/^/"/;s/$/"/\' | tr "\\n" ","`;\n                  export TF_CONFIG={\\"cluster\\":{\\"ps\\":[${PS_HOST}],\\"worker\\":[${WORKER_HOST}]},\\"task\\":{\\"type\\":\\"ps\\",\\"index\\":${VK_TASK_INDEX}},\\"environment\\":\\"cloud\\"};\n                  python /var/tf_dist_mnist/dist_mnist.py\n              image: volcanosh/dist-mnist-tf-example:0.0.1\n              name: tensorflow\n              ports:\n                - containerPort: 2222\n                  name: tfjob-port\n              resources: {}\n          restartPolicy: Never\n    - replicas: 2\n      name: worker\n      policies:\n        - event: TaskCompleted\n          action: CompleteJob\n      template:\n        spec:\n          containers:\n            - command:\n                - sh\n                - -c\n                - |\n                  PS_HOST=`cat /etc/volcano/ps.host | sed \'s/$/&:2222/g\' | sed \'s/^/"/;s/$/"/\' | tr "\\n" ","`;\n                  WORKER_HOST=`cat /etc/volcano/worker.host | sed \'s/$/&:2222/g\' | sed \'s/^/"/;s/$/"/\' | tr "\\n" ","`;\n                  export TF_CONFIG={\\"cluster\\":{\\"ps\\":[${PS_HOST}],\\"worker\\":[${WORKER_HOST}]},\\"task\\":{\\"type\\":\\"worker\\",\\"index\\":${VK_TASK_INDEX}},\\"environment\\":\\"cloud\\"};\n                  python /var/tf_dist_mnist/dist_mnist.py\n              image: volcanosh/dist-mnist-tf-example:0.0.1\n              name: tensorflow\n              ports:\n                - containerPort: 2222\n                  name: tfjob-port\n              resources: {}\n          restartPolicy: Never\n'})}),"\n",(0,t.jsxs)(e.p,{children:["Deploy ",(0,t.jsx)(e.code,{children:"tftest.yaml"}),"."]}),"\n",(0,t.jsx)(e.pre,{children:(0,t.jsx)(e.code,{children:"kubectl apply -f tftest.yaml\n"})}),"\n",(0,t.jsx)(e.p,{children:"View job health."}),"\n",(0,t.jsx)(e.pre,{children:(0,t.jsx)(e.code,{children:"kubectl get pod\n"})})]})}function p(n={}){const{wrapper:e}={...(0,r.R)(),...n.components};return e?(0,t.jsx)(e,{...n,children:(0,t.jsx)(d,{...n})}):d(n)}},84927(n,e,o){o.d(e,{A:()=>s});
1const s=o.p+"assets/images/ps-worker-8896dac5a82991f7a30d24980e3e50c5.png"},28453(n,e,o){o.d(e,{R:()=>i,x:()=>c});var s=o(96540);const t={},r=s.createContext(t);function i(n){const e=s.useContext(r);return s.useMemo(function(){return"function"==typeof n?n(e):{...e,...n}},[e,n])}function c(n){let e;return e=n.disableParentContext?"function"==typeof n.components?n.components(t):n.components||t:i(n.components),s.createElement(r.Provider,{value:e},n.children)}}}]);

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.