1"use strict";(self.webpackChunkcrawlee_website=self.webpackChunkcrawlee_website||[]).push([["10465"],{69814(e,r,s){s.r(r),s.d(r,{metadata:()=>t,default:()=>p,frontMatter:()=>a,contentTitle:()=>o,toc:()=>c,assets:()=>d});var t=JSON.parse('{"id":"guides/jsdom-crawler-guide","title":"JSDOMCrawler guide","description":"Your first steps into the world of scraping with Crawlee","source":"@site/versioned_docs/version-3.15/guides/jsdom_crawler.mdx","sourceDirName":"guides","slug":"/guides/jsdom-crawler-guide","permalink":"/js/docs/3.15/guides/jsdom-crawler-guide","draft":false,"unlisted":false,"editUrl":"https://github.com/apify/crawlee/edit/master/website/versioned_docs/version-3.15/guides/jsdom_crawler.mdx","tags":[],"version":"3.15","lastUpdatedBy":"Muhammad Haadhee Sheeraz Mian","lastUpdatedAt":1790948334000,"frontMatter":{"id":"jsdom-crawler-guide","title":"JSDOMCrawler guide","sidebar_label":"JSDOMCrawler","description":"Your first steps into the world of scraping with Crawlee"},"sidebar":"docs","previous":{"title":"Avoid getting blocked","permalink":"/js/docs/3.15/guides/avoid-blocking"},"next":{"title":"Impit HTTP Client","permalink":"/js/docs/3.15/guides/impit-http-client/"}}'),n=s(62540),i=s(43023),l=s(20188);let a={id:"jsdom-crawler-guide",title:"JSDOMCrawler guide",sidebar_label:"JSDOMCrawler",description:"Your first steps into the world of scraping with Crawlee"},o,d={},c=[{value:"How the crawler works",id:"how-the-crawler-works",level:2},{value:"When to use <code>JSDOMCrawler</code>
1",id:"when-to-use-jsdomcrawler",level:2},{value:"Example use of Element API",id:"example-use-of-element-api",level:2},{value:"Find all links on a page",id:"find-all-links-on-a-page",level:3},{value:"Other examples",id:"other-examples",level:3}];function h(e){let r={a:"a",admonition:"admonition",code:"code",h2:"h2",h3:"h3",li:"li",p:"p",pre:"pre",strong:"strong",ul:"ul",...(0,i.R)(),...e.components};return(0,n.jsxs)(n.Fragment,{children:[(0,n.jsxs)(r.p,{children:["\u200B",(0,n.jsx)(l.A,{to:"jsdom-crawler/class/JSDOMCrawler",children:(0,n.jsx)(r.code,{children:"JSDOMCrawler"})})," is very useful for scraping with the Window API."]}),"\n",(0,n.jsx)(r.h2,{id:"how-the-crawler-works",children:"How the crawler works"}),"\n",(0,n.jsxs)(r.p,{children:["\u200B",(0,n.jsx)(l.A,{to:"jsdom-crawler/class/JSDOMCrawler",children:(0,n.jsx)(r.code,{children:"JSDOMCrawler"})})," crawls by making plain HTTP requests to the provided URLs using the specialized ",(0,n.jsx)(r.a,{href:"https://github.com/apify/got-scraping",target:"_blank",rel:"noopener",children:"got-scraping"})," HTTP client. The URLs are fed to the crawler using ",(0,n.jsx)(l.A,{to:"core/class/RequestQueue",children:(0,n.jsx)(r.code,{children:"RequestQueue"})}),". The HTTP responses it gets back are usually HTML pages. The same pages you would get in your browser when you first load a URL. But it can handle any content types with the help of the ",(0,n.jsx)(l.A,{to:"jsdom-crawler/interface/JSDOMCrawlerOptions#additionalMimeTypes",children:(0,n.jsx)(r.code,{children:"additionalMimeTypes"})})," option."]}),"\n",(0,n.jsx)(r.admonition,{type:"info",children:(0,n.jsxs)(r.p,{children:["Modern web pages often do not serve all of their content in the first HTML response, but rather the first HTML contains links to other resources such as CSS and JavaScript that get downloaded afterwards, and together they create the final page. To crawl those, see ",(0,n.jsx)(l.A,{to:"puppeteer-crawler/class/PuppeteerCrawler",children:(0,n.jsx)(r.code,{children:"PuppeteerCrawler"})})," and ",(0,n.jsx)(l.A,{to:"playwright-crawler/class/PlaywrightCrawler",children:(0,n.jsx)(r.code,{children:"PlaywrightCrawler"})}),"."]})}),"\n",(0,n.jsxs)(r.p,{children:["Once the page's HTML is retrieved, the crawler will pass it to ",(0,n.jsx)(r.a,{href:"https://www.npmjs.com/package/jsdom",target:"_blank",rel:"noopener",children:"JSDOM"})," for parsing. The result is a ",(0,n.jsx)(r.code,{children:"window"})," property, which should be familiar to frontend developers. You can use the Window API to do all sorts of lookups and manipulation of the page's HTML, but in scraping, you will mostly use it to find specific HTML elements and extract their data."]}),"\n",(0,n.jsx)(r.p,{children:"Example use of browser JavaScript:"}),"\n",(0,n.jsx)(r.pre,{children:(0,n.jsx)(r.code,{className:"language-ts",children:"// Return the page title\ndocument.title; // browsers\nwindow.document.title; // JSDOM\n"})}),"\n",(0,n.jsxs)(r.h2,{id:"when-to-use-jsdomcrawler",children:["When to use ",(0,n.jsx)(r.code,{children:"JSDOMCrawler"})]}),"\n",(0,n.jsxs)(r.p,{children:[(0,n.jsx)(r.code,{children:"JSDOMCrawler"})," really shines when ",(0,n.jsx)(r.code,{children:"CheerioCrawler"})," is just not enough. There is an entire set of ",(0,n.jsx)(r.a,{href:"https://developer.mozilla.org/en-US/docs/Web/API/HTML_DOM_API",target:"_blank",rel:"noopener",children:"APIs"})," available!"]}),"\n",(0,n.jsx)(r.p,{children:(0,n.jsx)(r.strong,{children:"Advantages:"})}),"\n",(0,n.jsxs)(r.ul,{children:["\n",(0,n.jsx)(r.li,{children:"Easy to set up"}),"\n",(0,n.jsx)(r.li,{children:"Familiar for frontend developers"}),"\n",(0,n.jsx)(r.li,{children:"Content can be manipulated"}),"\n",(0,n.jsx)(r.li,{children:"Automatically avoids some anti-scraping bans"}),"\n"]}),"\n",(0,n.jsx)(r.p,{children:(0,n.jsx)(r.strong,{children:"Disadvantages:"})}),"\n",(0,n.jsxs)(r.ul,{children:["\n",(0,n.jsxs)(r.li,{children:["Slower than ",(0,n.jsx)(r.code,{children:"CheerioCrawler"})]}),"\n",(0,n.jsx)(r.li,{children:"Does not work for websites that require JavaScript rendering"}),"\n",(0,n.jsx)(r.li,{children:"May easily overload the target website with requests"}),"\n"]}),"\n",(0,n.jsx)(r.h2,{id:"example-use-of-element-api",children:"Example use of Element API"}),"\n",(0,n.jsx)(r.h3,{id:"find-all-links-on-a-page",children:"Find all links on a page"}),"\n",(0,n.jsxs)(r.p,{children:["This snippet finds all ",(0,n.jsx)(r.code,{children:"<a>"})," elements which have the ",(0,n.jsx)(r.code,{children:"href"})," attribute and extracts the hrefs into an array."]}),"\n",(0,n.jsx)(r.pre,{children:(0,n.jsx)(r.code,{className:"language-js",children:"Array.from(document.querySelectorAll('a[href]')).map((a) => a.href);\n"})}),"\n",(0,n.jsx)(r.h3,{id:"other-examples",children:"Other examples"}),"\n",(0,n.jsxs)(r.p,{children:["Visit the ",(0,n.jsx)(r.a,{href:"../examples",children:"Examples"})," section to browse examples of ",(0,n.jsx)(r.code,{children:"JSDOMCrawler"})," usage. Almost all examples show ",(0,n.jsx)(r.code,{children:"JSDOMCrawler"})," code in their code tabs."]})]})}function p(e={}){let{wrapper:r}={...(0,i.R)(),...e.components};return r?(0,n.jsx)(r,{...e,children:(0,n.jsx)(h,{...e})}):h(e)}},20188(e,r,s){s.d(r,{A:()=>
1o});var t=s(62540);s(63696);var n=s(12212),i=s(17607),l=s(73228);let a=s(6715)[0],o=({to:e,children:r})=>{let s=(0,i.r)(),{siteConfig:o}=(0,l.A)();return o.presets[0][1].docs.disableVersioning||s.version===a?(0,t.jsx)(n.A,{to:`/js/api/${e}`,children:r}):(0,t.jsx)(n.A,{to:`/js/api/${"current"===s.version?"next":s.version}/${e}`,children:r})}},43023(e,r,s){s.d(r,{R:()=>l,x:()=>a});var t=s(63696);let n={},i=t.createContext(n);function l(e){let r=t.useContext(i);return t.useMemo(function(){return"function"==typeof e?e(r):{...r,...e}},[r,e])}function a(e){let r;return r=e.disableParentContext?"function"==typeof e.components?e.components(n):e.components||n:l(e.components),t.createElement(i.Provider,{value:r},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.