PageSourceSearch

https://www.bsc.es/supportkc/assets/js/f8dd1e30.e0fe03ac.js

js bsc.es collected 2026-10-01 09:25:00 UTC 17,537 bytes, 1 lines download raw bytes

1"use strict";(self.webpackChunksupportkc_new=self.webpackChunksupportkc_new||[]).push([[1715],{3905:(e,t,n)=>{n.d(t,{Zo:()=>u,kt:()=>g});var a=n(67294);function r(e,t,n){return t in e?Object.defineProperty(e,t,{value:n,enumerable:!0,configurable:!0,writable:!0}):e[t]=n,e}function i(e,t){var n=Object.keys(e);if(Object.getOwnPropertySymbols){var a=Object.getOwnPropertySymbols(e);t&&(a=a.filter((function(t){return Object.getOwnPropertyDescriptor(e,t).enumerable}))),n.push.apply(n,a)}return n}function o(e){for(var t=1;t<arguments.length;t++){var n=null!=arguments[t]?arguments[t]:{};t%2?i(Object(n),!0).forEach((function(t){r(e,t,n[t])})):Object.getOwnPropertyDescriptors?Object.defineProperties(e,Object.getOwnPropertyDescriptors(n)):i(Object(n)).forEach((function(t){Object.defineProperty(e,t,Object.getOwnPropertyDescriptor(n,t))}))}return e}function l(e,t){if(null==e)return{};var n,a,r=function(e,t){if(null==e)return{};var n,a,r={},i=Object.keys(e);for(a=0;a<i.length;a++)n=i[a],t.indexOf(n)>=0||(r[n]=e[n]);return r}(e,t);if(Object.getOwnPropertySymbols){var i=Object.getOwnPropertySymbols(e);for(a=0;a<i.length;a++)n=i[a],t.indexOf(n)>=0||Object.prototype.propertyIsEnumerable.call(e,n)&&(r[n]=e[n])}return r}var s=a.createContext({}),p=function(e){var t=a.useContext(s),n=t;return e&&(n="function"==typeof e?e(t):o(o({},t),e)),n},u=function(e){var t=p(e.components);return a.createElement(s.Provider,{value:t},e.children)},c="mdxType",d={inlineCode:"code",wrapper:function(e){var t=e.children;return a.createElement(a.Fragment,{},t)}},m=a.forwardRef((function(e,t){var n=e.components,r=e.mdxType,i=e.originalType,s=e.parentName,u=l(e,["components","mdxType","originalType","parentName"]),c=p(n),m=r,g=c["".concat(s,".").concat(m)]||c[m]||d[m]||i;return n?a.createElement(g,o(o({ref:t},u),{},{components:n})):a.createElement(g,o({ref:t},u))}));function g(e,t){var n=arguments,r=t&&t.mdxType;if("string"==typeof e||r){var i=n.length,o=new Array(i);o[0]=m;var l={};for(var s in t)hasOwnProperty.call(t,s)&&(l[s]=t[s]);l.originalType=e,l[c]="string"==typeof e?e:r,o[1]=l;for(var p=2;p<i;p++)o[p]=n[p];return a.createElement.apply(null,o)}return a.createElement.apply(null,n)}m.displayName="MDXCreateElement"},88552:(e,t,n)=>{n.r(t),n.d(t,{assets:()=>s,contentTitle:()=>o,default:()=>d,frontMatter:()=>i,metadata:()=>l,toc:()=>p});var a=n(87462),r=(n(67294),n(3905));const i={sidebar_position:7,sidebar_label:"EAR"},o="MareNostrum 5",l={unversionedId:"MareNostrum5/EAR",id:"MareNostrum5/EAR",title:"MareNostrum 5",description:"EAR",source:"@site/docs/MareNostrum5/EAR.md",sourceDirName:"MareNostrum5",slug:"/MareNostrum5/EAR",permalink:"/supportkc/docs/MareNostrum5/EAR",draft:!1,tags:[],version:"current",sidebarPosition:7,frontMatter:{sidebar_position:7,sidebar_label:"EAR"},sidebar:"tutorialSidebar",previous:{title:"Hardware Counters",permalink:"/supportkc/docs/MareNostrum5/profilingperf"},next:{title:"Installing packages",permalink:"/supportkc/docs/MareNostrum5/Installing packages/install_pkgs_menu"}},s={},p=[{value:"EAR",id:"ear",level:2},{value:"Running jobs with EAR",id:"running-jobs-with-ear",level:3},{value:"Gathering Data",id:"gathering-data",level:3},{value:"Data visualization",id:"data-visualization",level:3},{value:"CPU and GPU Frequency Selection",id:"cpu-and-gpu-frequency-selection",level:3}],u={toc:p},c="wrapper";function d(e){let{components:t,...i}=e;return(0,r.kt)(c,(0,a.Z)({},u,i,{components:t,mdxType:"MDXLayout"}),(0,r.kt)("h1",{id:"marenostrum-5"},"MareNostrum 5"),(0,r.kt)("h2",{id:"ear"},"EAR"),(0,r.kt)("p",null,(0,r.kt)("strong",{parentName:"p"},"EAR")," (Energy Aware Runtime) is a management framework optimizing the energy and efficiency of a cluster of interconnected nodes. To improve the energy of the cluster, EAR provides energy control, accounting, monitoring and optimization of both the applications running on the cluster and of the overall global cluster."),(0,r.kt)("p",null,"The EAR SLURM plug-in allows applications to be executed with EAR using standard commands such as srun, sbatch, or mpirun. When EAR is enabled, the EAR Library (EARL) is automatically loaded for supported applications. On MN5, EAR v4.3 is available in both GPP and ACC partitions. The EAR modules available in each partition can be listed with:"),(0,r.kt)("pre",null,(0,r.kt)("code",{parentName:"pre",className:"language-bash"},"module avail ear\n")),(0,r.kt)("p",null,"To see all EAR-related options that can be used at job submission time, run:"),(0,r.kt)("pre",null,(0,r.kt)("code",{parentName:"pre",className:"language-bash"},"sbatch --help\n")),(0,r.kt)("h3",{id:"running-jobs-with-ear"},"Running jobs with EAR"),(0,r.kt)("p",null,"EAR collects performance and energy metrics through EARL. Metrics can be accessed after the job finishes using the ",(0,r.kt)("strong",{parentName:"p"},"eacct")," tool, which retrieves job, per-node, or per-loop data from the EAR database. Alternatively, runtime reporting can write metrics to CSV files while the job is running, enabled via the  ",(0,r.kt)("inlineCode",{parentName:"p"},"--ear-user-db")," flag or by setting SLURM_EAR_REPORT_ADD=csv_ts.so."),
1(0,r.kt)("p",null,"The following EAR options can be specified when running srun and/or sbatch, and are supported with\nsrun/sbatch/salloc:"),(0,r.kt)("table",null,(0,r.kt)("thead",{parentName:"table"},(0,r.kt)("tr",{parentName:"thead"},(0,r.kt)("th",{parentName:"tr",align:null},"Option"),(0,r.kt)("th",{parentName:"tr",align:null},"Description"))),(0,r.kt)("tbody",{parentName:"table"},(0,r.kt)("tr",{parentName:"tbody"},(0,r.kt)("td",{parentName:"tr",align:null},"--ear=","[on-off]"),(0,r.kt)("td",{parentName:"tr",align:null},"Enables/disables EAR library loading with this job.")),(0,r.kt)("tr",{parentName:"tbody"},(0,r.kt)("td",{parentName:"tr",align:null},"--ear-user-db=filename"),(0,r.kt)("td",{parentName:"tr",align:null},"Asks the EAR Library to generate a set of CSV files with EARL metrics.")),(0,r.kt)("tr",{parentName:"tbody"},(0,r.kt)("td",{parentName:"tr",align:null},"--ear-verbose=","[0-1]"),(0,r.kt)("td",{parentName:"tr",align:null},"Specifies the level of verbosity; the default is 0.")))),(0,r.kt)("p",null,"When using ",(0,r.kt)("inlineCode",{parentName:"p"}
1,"--ear-user-db")," flag, one file per node is generated with the average node metrics (node signature)\nand one file with multiple lines per node is generated with runtime collected metrics (loops node signatures)."),(0,r.kt)("p",null,"EAR supports a variety of use cases, including MPI applications compiled with IntelMPI or OpenMPI, non-MPI applications (CUDA, OpenMP, MKL), Python MPI applications, Singularity containers, and others. You can find a complete list of supported use cases in the ",(0,r.kt)("a",{parentName:"p",href:"https://gitlab.bsc.es/ear_team/ear/-/wikis/User-guide#use-cases"},"EAR documentation"),". The following example shows a ",(0,r.kt)("strong",{parentName:"p"},"SLURM job script")," running ",(0,r.kt)("a",{parentName:"p",href:"/docs/MareNostrum5/Marenostrum5-Applications/ACC/GROMACS"},"GROMACS")," compiled with OpenMPI on the ACC partition with EAR enabled:"),(0,r.kt)("pre",null,(0,r.kt)("code",{parentName:"pre"},'#!/bin/bash\n#------------------------------------------------------\n# Example SLURM job script with SBATCH requesting GPUs\n#------------------------------------------------------\n#SBATCH --job-name=gromacs\n#SBATCH --account=bsc99\n#SBATCH --qos=acc_bench\n#SBATCH -o outs/slurm_output.%j\n#SBATCH -e outs/slurm_error.%j\n#SBATCH --nodes=1\n#SBATCH --ntasks-per-node=8\n#SBATCH --cpus-per-task=8\n#SBATCH --time=02:00:00\n#SBATCH --exclusive\n#SBATCH --gres=gpu:4\n#SBATCH --constraint=perfparanoid\n#SBATCH --ear=on\n#SBATCH --ear-user-db=metrics/gromacs_metrics\n\nmodule purge\nmodule load nvidia-hpc-sdk\nmodule load fftw/3.3.10-gcc-nvhpcx\nmodule load gromacs/2024.2\nmodule load ear\n\n\nexport EARL_REPORT_LOOPS=1\nexport SLURM_CPU_BIND=none\n\n\nmpirun -npernode 8  erun --program="gmx_mpi mdrun -ntomp 8 -nb gpu -pme gpu -npme 1 -update gpu -bonded gpu -nsteps 100000 -resetstep 90000 -noconfout -dlb no -nstlist 300 -pin on -v -gpu_id 0123"\n')),(0,r.kt)("admonition",{title:"Important",type:"caution"},(0,r.kt)("p",{parentName:"admonition"},"When using EAR, it is mandatory to include the SLURM directive #SBATCH --constraint=perfparanoid in your job script. Once this is set, the EAR counters will become available.")),(0,r.kt)("h3",{id:"gathering-data"},"Gathering Data"),(0,r.kt)("p",null,"Since EAR v4.3 is officially installed, job data can be retrieved using the ",(0,r.kt)("strong",{parentName:"p"},"eacct")," command. To use this tool, the ear module must be loaded beforehand."),(0,r.kt)("pre",null,(0,r.kt)("code",{parentName:"pre",className:"language-bash"},"module load ear\n")),(0,r.kt)("p",null,"For example, the following command shows metrics for job ID 33959523, corresponding to a GROMACS application that was run on a single node in the ACC partition:"),(0,r.kt)("pre",null,(0,r.kt)("code",{parentName:"pre",className:"language-sbatch"},"eacct -j 33959523\n   JOB-STEP USER       APPLICATION      POLICY NODES AVG/DEF/IMC(GHz) TIME(s)    POWER(W) GBS     CPI   ENERGY(J)    GFLOPS/W IO(MBs) MPI%  G-POW (T/U)     G-FREQ  G-UTIL(G/MEM)\n33959523-sb   bsc099998  gromacs          NP     1     3.23/2.00/---    288.00     1569.22  ---     ---   451935       ---      ---     ---   ---             ---     ---          \n33959523-0    bsc099998  gmx_mpi          MO     1     3.31/2.00/2.40   282.03     1622.34  4.15    0.27  457556       0.0007   0.9     56.4  1102.06/1102.06 1.975   73%/4% \n")),(0,r.kt)("p",null,"By default, eacct reports a pre-selected set of aggregated metrics. To obtain more detailed information, the ",(0,r.kt)("inlineCode",{parentName:"p"},"-l flag")," provides node-level accounting, while the ",(0,r.kt)("inlineCode",{parentName:"p"},"-r glag")," retrieves runtime metrics collected during execution (EAR loops), if EARL was loaded for the job. "),(0,r.kt)("p",null,"To gather all available metrics collected by EAR and facilitate post-processing, the ",(0,r.kt)("inlineCode",{parentName:"p"},"-c flag")," can be used to export the data in CSV format. For example:"),(0,r.kt)("pre",null,(0,r.kt)("code",{parentName:"pre",className:"language-sbatch"},"eacct -j 33959523 -c gromacs_app.csv \n\nSuccessfully written applications to csv. Only applications with EARL will have its information properly written.\n")),(0,r.kt)("p",null,"These metrics provide key information on job performance, resource usage, and energy efficie
1ncy. The most commonly used metrics include the following:"),(0,r.kt)("table",null,(0,r.kt)("thead",{parentName:"table"},(0,r.kt)("tr",{parentName:"thead"},(0,r.kt)("th",{parentName:"tr",align:null},"Metric"),(0,r.kt)("th",{parentName:"tr",align:null},"Description"))),(0,r.kt)("tbody",{parentName:"table"},(0,r.kt)("tr",{parentName:"tbody"},(0,r.kt)("td",{parentName:"tr",align:null},"TIME(s)"),(0,r.kt)("td",{parentName:"tr",align:null},"Step execution time, in seconds.")),(0,r.kt)("tr",{parentName:"tbody"},(0,r.kt)("td",{parentName:"tr",align:null},"AVG/DEF/IMC(GHz)"),(0,r.kt)("td",{parentName:"tr",align:null},"CPU frequencies: average, default, and uncore (IMC) frequency.")),(0,r.kt)("tr",{parentName:"tbody"},(0,r.kt)("td",{parentName:"tr",align:null},"CPI"),(0,r.kt)("td",{parentName:"tr",align:null},"Cycles per instruction, another indicator of CPU/memory performance.")),(0,r.kt)("tr",{parentName:"tbody"},(0,r.kt)("td",{parentName:"tr",align:null},"POWER(W)"),(0,r.kt)("td",{parentName:"tr",align:null},"Average node power consumption in watts.")),(0,r.kt)("tr",{parentName:"tbody"},(0,r.kt)("td",{parentName:"tr",align:null},"ENERGY(J)"),(0,r.kt)("td",{parentName:"tr",align:null},"Total energy consumed by the node(s) in joules.")),(0,r.kt)("tr",{parentName:"tbody"},(0,r.kt)("td",{parentName:"tr",align:null},"MPI%"),(0,r.kt)("td",{parentName:"tr",align:null},"Percentage of total time spent in MPI calls.")),(0,r.kt)("tr",{parentName:"tbody"},(0,r.kt)("td",{parentName:"tr",align:null},"IO(MBs)"),(0,r.kt)("td",{parentName:"tr",align:null},"IO (read and write) Mega Bytes per second.")))),(0,r.kt)("h3",{id:"data-visualization"},"Data visualization"),(0,r.kt)("p",null,(0,r.kt)("strong",{parentName:"p"},"ear-job-visualizer")," is a Python-based CLI tool that reads runtime data generated by the EAR software and displays EAR metrics as timeline plots. It allows users to visualize how performance and energy-related metrics (e.g., GFLOPS, MPI percentage, I/O bandwidth, GPU utilization, or power) evolve during the execution of a job."),(0,r.kt)("p",null,"To use the tool, the miniforge module must be loaded:"),(0,r.kt)("pre",null,(0,r.kt)("code",{parentName:"pre",className:"language-bash"},"module load miniforge\n")),(0,r.kt)("p",null,"and the corresponding environment activated:"),(0,r.kt)("pre",null,(0,r.kt)("code",{parentName:"pre",className:"language-bash"},"source activate ear-job-visualization\n")),(0,r.kt)("p",null,"To generate the CSV files required by the visualizer, the job must first be submitted with the ",(0,r.kt)("inlineCode",{parentName:"p"}," --ear-user-db"),"  option enabled. EAR produces separate CSV files for application-level (",(0,r.kt)("em",{parentName:"p"},".time.csv) and loop-level ("),".time.loops.csv) metrics, which should be placed in different directories and passed to the tool through the ",(0,r.kt)("inlineCode",{parentName:"p"},"--apps-file")," and ",(0,r.kt)("inlineCode",{parentName:"p"},"--loops-file")," parameter. ",(0,r.kt)("strong",{parentName:"p"},"Runtime visualizations")," are generated using the ",(0,r.kt)("inlineCode",{parentName:"p"},"--format "),"runtime option, specifying the job and step identifiers and selecting the desired metrics with ",(0,r.kt)("inlineCode",{parentName:"p"},"-m"),". "),(0,r.kt)("p",null,"For example, the following command generates timeline plots for I/O bandwidth, floating-point performance, and MPI percentage:"),(0,r.kt)("pre",null,(0,r.kt)("code",{parentName:"pre",className:"language-bash"},"ear-job-visualizer --format runtime --job-id 36290942 --step-id 0  --loops-file loops_dir --apps-file apps_dir -m io_mbs gflops perc_mpi\n")),(0,r.kt)("p",null,"The tool generates the following runtime figures: runtime_perc_mpi.png, runtime_io_mbs.png and runtime_gflops.png. As an example, the following figure corresponds to runtime_perc_mpi.png:"),(0,r.kt)("p",null,(0,r.kt)("img",{alt:"MPI runtime",src:n(96175).Z,width:"754",height:"195"})),(0,r.kt)("p",null,"This image represents the percentage of time spent in MPI throughout the execution of the job step. The horizontal axis corresponds to the runtime of the application, while the color scale indicates the MPI percentage (%MPI). Each segment along the timeline reflects the communication intensity during that interval. Higher values (towards the upper end of the color bar) indicate phases where the application is more communication-bound, whereas lower values correspond to computation-dominated regi
1ons. This type of visualization helps identify communication-heavy phases and assess the impact of MPI on overall performance."),(0,r.kt)("p",null,"The tool can also be used to visualize GPU-related metrics, such as utilization and power. For instance, the following command generates timeline plots for GPU utilization and GPU power:"),(0,r.kt)("pre",null,(0,r.kt)("code",{parentName:"pre",className:"language-bash"},"ear-job-visualizer --format runtime --job-id 36290942 --step-id 0 --loops-file loops_dir --apps-file apps_dir -m gpu_util gpu_power\n")),(0,r.kt)("p",null,"The following figure shows the GPU utilization percentage for all four GPUs in the node throughout the execution of the job step:"),(0,r.kt)("p",null,(0,r.kt)("img",{alt:"GPU utilization",src:n(64907).Z,width:"796",height:"343"})),(0,r.kt)("p",null,"Each line represents one GPU, showing periods of high or low activity. "),(0,r.kt)("p",null,"In addition, EAR Job Visualizer can generate Paraver traces using the ",(0,r.kt)("inlineCode",{parentName:"p"},"--format ear2prv")," option. For example:"),(0,r.kt)("pre",null,(0,r.kt)("code",{parentName:"pre",className:"language-bash"},"ear-job-visualizer --format ear2prv --job-id 36290942 --step-id 0  --loops-file loops_dir --apps-file apps_dir -m io_mbs gflops perc_mpi\n")),(0,r.kt)("h3",{id:"cpu-and-gpu-frequency-selection"},"CPU and GPU Frequency Selection"),(0,r.kt)("p",null,"EAR allows authorized users to request specific CPU and GPU frequencies for their jobs. The CPU frequency can be requested using\nthe ",(0,r.kt)("inlineCode",{parentName:"p"},"--ear-cpufreq=value")," flag (value in kHz), and the desired optimization policy can be selected with ",(0,r.kt)("inlineCode",{parentName:"p"},"--ear-policy=policy_name"),". Type srun --help to see the policies currently installed on your system. Contact the system administrator or helpdesk team to become an authorized user."),(0,r.kt)("p",null,"For GPU-enabled jobs, EAR supports GPU monitoring for NVIDIA devices. Although GPU frequency optimization is not fully supported yet, authorized users can request a specific GPU frequency for all GPUs on a node by setting the ",(0,r.kt)("inlineCode",{parentName:"p"},"SLURM_EAR_GPU_DEF_FREQ")," environment variable (value in kHz)"))}d.isMDXComponent=!0},64907:(e,t,n)=>{n.d(t,{Z:()=>a});const a=n.p+"assets/images/runtime_gpu_util-8d8c7e94a5e24229b2243d8a857a5fda.png"},96175:(e,t,n)=>{n.d(t,{Z:()=>a});const a=n.p+"assets/images/runtime_perc_mpi-776843f48df3691a223af854379d8264.png"}}]);

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.