1"use strict";(globalThis.webpackChunklanding_blog=globalThis.webpackChunklanding_blog||[]).push([[1267],{68557(e,n,t){t.r(n),t.d(n,{assets:()=>c,contentTitle:()=>a,default:()=>h,frontMatter:()=>l,metadata:()=>r,toc:()=>u});var r=t(27390),s=t(74848),o=t(28453),i=t(86025);const l={slug:"open-source-scraping-with-million-of-browsers-or-puppeteer-cluster",title:"Scraping with millions of browsers or Puppeteer Cluster",description:"Check out how to run open-source scraping with million of browsers using Puppeteer Cluster. Run your own pool of Chromium instances controlled by Puppeteer.",author:"Oleg Kulyk",author_title:"Co-Founder @ ScrapingAnt",author_url:"https://www.linkedin.com/in/kami4ka/",author_image_url:"https://avatars0.githubusercontent.com/u/5595029?s=400&v=4",image:"/img/blog/open-source-scraping-with-million.jpg",tags:["web scraping","data extraction","javascript"]},a=void 0,c={authorsImageUrls:[void 0]},u=[{value:"Running a pool of Chromium instances using Puppeteer",id:"running-a-pool-of-chromium-instances-using-puppeteer",level:2},{value:"Installing Puppeteer Cluster",id:"installing-puppeteer-cluster",level:3},{value:"Usage",id:"usage",level:3},{value:"Documentation",id:"documentation",level:3},{value:"Conclusion",id:"conclusion",level:2}];function p(e){const n={a:"a",code:"code",h2:"h2",h3:"h3",img:"img",li:"li",p:"p",pre:"pre",ul:"ul",...(0,o.R)(),...e.components};return(0,s.jsxs)(s.Fragment,{children:[(0,s.jsx)(n.p,{children:(0,s.jsx)(n.img,{alt:"Scraping with millions of browsers or Puppeteer Cluster",src:t(98385).A+"",width:"640",height:"397"})}),"\n",(0,s.jsx)(n.p,{children:"In this article, we\u2019d like to introduce an awesome open-source Web Scraping solution for running a pool of Chromium instances using Puppeteer."}),"\n",(0,s.jsx)(n.h2,{id:"running-a-pool-of-chromium-instances-using-puppeteer",children:"Running a pool of Chromium instances using Puppeteer"}),"\n",(0,s.jsx)(n.p,{children:"Sequential execution to perform web scraping tasks is not a good idea as one process has to wait for the other processes to complete first. This is a time-consuming job when it comes to many processes waiting in a queue. So to overcome this we are going to perform these actions in parallel where the processes execute concurrently and result in less time consumption."}),"\n",(0,s.jsx)(n.p,{children:"What does this library do?"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"Handling of crawling errors"}),"\n",(0,s.jsx)(n.li,{children:"Auto restarts the browser in case of a crash"}),"\n",(0,s.jsx)(n.li,{children:"Can automatically retry if a job fails"}),"\n",(0,s.jsx)(n.li,{children:"Different concurrency models to choose from (pages, contexts, browsers)"}),"\n",(0,s.jsx)(n.li,{children:"Simple to use, small boilerplate"}),"\n",(0,s.jsx)(n.li,{children:"Progress view and monitoring statistics (see below)"}),"\n"]}),"\n",(0,s.jsx)("p",{align:"center",children:(0,s.jsx)("img",{alt:"Cluster example",src:(0,i.Ay)("img/blog/cluster.gif"),width:"90%"})}),"\n",(0,s.jsxs)(n.p,{children:["To read more about Puppeteer itself just visit the official Github page: ",(0,s.jsx)(n.a,{href:"https://github.com/puppeteer/puppeteer",children:"https://github.com/puppeteer/puppeteer"})]}),"\n",(0,s.jsx)(n.h3,{id:"installing-puppeteer-cluster",children:"Installing Puppeteer Cluster"}),"\n",(0,s.jsxs)(n.p,{children:["To start using Puppeteer Cluster you should start by installing dependencies, for example, via ",(0,s.jsx)(n.code,{children:"NPM"}),"."]}),"\n",(0,s.jsxs)(n.p,{children:["Install ",(0,s.jsx)(n.code,{children:"puppeteer"})," (if you don't already have it installed):"]}),"\n",(0,s.jsx)(n.pre,{children:(0,s.jsx)(n.code,{className:"language-shell",children:"npm install --save puppeteer\n"})}),"\n",(0,s.jsxs)(n.p,{children:["Then install ",(0,s.jsx)(n.code,{children:"puppeteer-cluster"}),":"]}),"\n",(0,s.jsx)(n.pre,{children:(0,s.jsx)(n.code,{className:"language-shell",children:"npm install --save puppeteer-cluster\n"})}),"\n",(0,s.jsx)(n.h3,{id:"usage",children:"Usage"}),"\n",(0,s.jsx)(n.p,{children:"All that you need to provide while using Puppeteer Cluster function is:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"Process count that needs to be executed in parallel"}),"\n"]}),"\n",(0,s.jsx)(n.pre,{children:(0,s.jsx)(n.code,{className:"language-js",children:"const cluster = await Cluster.launch({\n concurrency: Cluster.CONCURRENCY_CONTEXT,\n maxConcurrency: 2 // Can be 1,000,00 but let's start from 2\n});\n"})}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"Define a task that has to perform the expected action"}),"\n"]}),"\n",(0,s.jsx)(n.pre,{children:(0,s.jsx)(n.code,{className:"language-js",children:"await cluster.task(async ({ page, data: url }) =>
1 {\n await page.goto(url);\n //Perform the action, for example - store the result\n})\n"})}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"Invoke the task using queue and wait until the cluster completes the execution"}),"\n"]}),"\n",(0,s.jsx)(n.pre,{children:(0,s.jsx)(n.code,{className:"language-js",children:"cluster.queue('http://www.google.com/');\ncluster.queue('https://scrapingant.com/');\n"})}),"\n",(0,s.jsx)(n.p,{children:"The whole script will look like the below example:"}),"\n",(0,s.jsx)(n.pre,{children:(0,s.jsx)(n.code,{className:"language-js",children:"const { Cluster } = require('puppeteer-cluster');\n\n(async () => {\n const cluster = await Cluster.launch({\n concurrency: Cluster.CONCURRENCY_CONTEXT,\n maxConcurrency: 2,\n });\n\n await cluster.task(async ({ page, data: url }) => {\n await page.goto(url);\n //Perform the action, for example - store the result\n });\n\n cluster.queue('http://www.google.com/');\n cluster.queue('https://scrapingant.com/');\n // many more pages\n\n await cluster.idle();\n await cluster.close();\n})();\n"})}),"\n",(0,s.jsx)(n.p,{children:"And that is all."}),"\n",(0,s.jsx)(n.h3,{id:"documentation",children:"Documentation"}),"\n",(0,s.jsxs)(n.p,{children:["For the extensive documentation just visit the Github repository: ",(0,s.jsx)(n.a,{href:"https://github.com/thomasdondorf/puppeteer-cluster",children:"https://github.com/thomasdondorf/puppeteer-cluster"})]}),"\n",(0,s.jsxs)(n.p,{children:["And the examples directory inside the repository: ",(0,s.jsx)(n.a,{href:"https://github.com/thomasdondorf/puppeteer-cluster/tree/master/examples",children:"https://github.com/thomasdondorf/puppeteer-cluster/tree/master/examples"})]}),"\n",(0,s.jsx)(n.h2,{id:"conclusion",children:"Conclusion"}),"\n",(0,s.jsx)(n.p,{children:"Apart from scraping you can still make use of Puppeteer Cluster for automation testing, performance testing, improving your site rank, and many more cases with parallel browser workers."}),"\n",(0,s.jsxs)(n.p,{children:["Of course, you can try our ",(0,s.jsx)(n.a,{href:"https://scrapingant.com",children:"Web Scraping API"})," that supports parallel execution of headless Chrome rendering with a simple API."]})]})}function h(e={}){const{wrapper:n}={...(0,o.R)(),...e.components};return n?(0,s.jsx)(n,{...e,children:(0,s.jsx)(p,{...e})}):p(e)}},98385(e,n,t){t.d(n,{A:()=>r});const r=t.p+"assets/images/open-source-scraping-with-million-eef7e63effe750a20c84e72c1b7e4a6c.jpg"},28453(e,n,t){t.d(n,{R:()=>i,x:()=>l});var r=t(96540);const s={},o=r.createContext(s);function i(e){const n=r.useContext(o);return r.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function l(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(s):e.components||s:i(e.components),r.createElement(o.Provider,{value:n},e.children)}},27390(e){e.exports=JSON.parse('{"permalink":"/blog/open-source-scraping-with-million-of-browsers-or-puppeteer-cluster","source":"@site/blog/2020-07-14-open-source-scraping-with-million-of-browsers-or-puppeteer-cluster.md","title":"Scraping with millions of browsers or Puppeteer Cluster","description":"Check out how to run open-source scraping with million of browsers using Puppeteer Cluster. Run your own pool of Chromium instances controlled by Puppeteer.","date":"2020-07-14T00:00:00.000Z","tags":[{"inline":true,"label":"web scraping","permalink":"/blog/tags/web-scraping"},{"inline":true,"label":"data extraction","permalink":"/blog/tags/data-extraction"},{"inline":true,"label":"javascript","permalink":"/blog/tags/javascript"}],"readingTime":2.32,"hasTruncateMarker":true,"authors":[{"name":"Oleg Kulyk","title":"Co-Founder @ ScrapingAnt","url":"https://www.linkedin.com/in/kami4ka/","imageURL":"https://avatars0.githubusercontent.com/u/5595029?s=400&v=4","key":null,"page":null}],"frontMatter":{"slug":"open-source-scraping-with-million-of-browsers-or-puppeteer-cluster","title":"Scraping with millions of browsers or Puppeteer Cluster","description":"Check out how to run open-source scraping with million of browsers using Puppeteer Cluster. Run your own pool of Chromium instances controlled by Puppeteer.","author":"Oleg Kulyk","author_title":"Co-Founder @ ScrapingAnt","author_url":"https://www.linkedin.com/in/kami4ka/","author_image_url":"https://avatars0.githubusercontent.com/u/5595029?s=400&v=4","image":"/img/blog/open-source-scraping-with-million.jpg","tags":["web scraping","data extraction","javascript"]},"unlisted":false,"lastUpdatedAt":null,"lastUpdatedBy":null,"prevItem":{"title":"Open Source Javascript Web Scraping","permalink":"/blog/awesome-open-source-javascript-projects-for-web-scraping"},"nextItem":{"title":"How to run Playwright on AWS Lambda","permalink":"/blog/how-to-run-playwright-on-aws-lambda"}}')}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.