PageSourceSearch

https://davehudson.io/blog/2024-07-15/code1.js

js davehudson.io collected 2026-10-04 07:39:25 UTC 2,042 bytes, 60 lines download raw bytes

1import puppeteer from 'puppeteer';
2import fs from 'fs';
3import path from 'path';
4import {fileURLToPath} from 'url';
5import {dirname} from 'path';
6import {parseStringPromise} from 'xml2js';
7
8// Get the current module path
9const __filename = fileURLToPath(import.meta.url);
10const __dirname = dirname(__filename);
11
12// Define the path to your sitemap.xml file
13const sitemapPath = path.join(__dirname, 'sitemap.xml');
14const outputDir = path.join(__dirname, 'prerendered');
15const localBaseUrl = 'http://localhost:3000';
16const maxConcurrentRenders = 4;
17
18// Utility function to create directories recursively
19const ensureDirectoryExistence = (filePath) => {
20    const dirname = path.dirname(filePath);
21    if (fs.existsSync(dirname)) {
22        return true;
23    }
24    ensureDirectoryExistence(dirname);
25    fs.mkdirSync(dirname);
26};
27
28(async () => {
29    // Read and parse the sitemap.xml file
30    const sitemapData = fs.readFileSync(sitemapPath, 'utf8');
31    const sitemap = await parseStringPromise(sitemapData);
32
33    // Extract URLs from the sitemap
34    const urls = sitemap.urlset.url.map(entry => entry.loc[0]);
35
36    // Launch Puppeteer
37    const browser = await puppeteer.launch();
38
39    // Helper function to render a single page
40    const renderPage = async (url) => {
41        const localUrl = url.replace(/^https?:\/\/[^\/]+/, localBaseUrl);
42        const page = await browser.newPage();
43        await page.goto(localUrl, {waitUntil: 'networkidle0'});
44        const html = await page.content();
45        const urlPath = new URL(url).pathname;
46        const filePath = path.join(outputDir, urlPath, 'index.html');
47        ensureDirectoryExistence(filePath);
48        fs.writeFileSync(filePath, html);
49        await page.close();
50    };
51
52    // Process URLs in batches of maxConcurrentRenders
53    for (let i = 0; i < urls.length; i += maxConcurrentRenders) {
54        const batch = urls.slice(i, i + maxConcurrentRenders).map(url => renderPage(url));
55        await Promise.all(batch);
56    }
57
58    // Close the browser
59    await browser.close();
60})();

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.