ScreenshotNeo

BlogHow-to

Use Node.js Playwright to Screenshot Every Page in a Sitemap

Build a Node.js Playwright script that expands sitemap indexes, captures listed URLs, and records failures with stable filenames.

By the ScreenshotNeo team4 October 202612 min read

To screenshot every page listed in a sitemap, fetch the sitemap XML, expand any sitemap indexes, deduplicate and validate the listed URLs, then capture each URL with Playwright. The script below saves viewport screenshots, uses bounded concurrency, writes a JSONL result for every URL, and continues after individual failures.

This covers the URLs in the sitemap files the script processes. A sitemap is an input list, not proof that it contains every reachable page on a site. A sitemap index points to other sitemap files, so it must be expanded to get its page entries. See the Sitemap Protocol and Google’s sitemap guidance.

1. Set up Node.js and Playwright

Create a project and install Playwright plus an XML parser. This example uses the fast-xml-parser package:

mkdir sitemap-shots
cd sitemap-shots
npm init -y
npm install playwright fast-xml-parser
npx playwright install chromium

Save the following as screenshot-sitemap.mjs. It uses Node’s built-in fetch, filesystem, path, crypto, and zlib modules, along with the installed packages. If your site requires authentication, add authorized cookies or headers to the browser context as described below.

2. Use this complete script

import { createWriteStream } from 'node:fs';
import { mkdir } from 'node:fs/promises';
import { createHash } from 'node:crypto';
import { gunzipSync } from 'node:zlib';
import path from 'node:path';
import { once } from 'node:events';
import { XMLParser } from 'fast-xml-parser';
import { chromium } from 'playwright';

const startUrl = process.argv[2];
if (!startUrl) {
  console.error('Usage: node screenshot-sitemap.mjs <sitemap-or-robots-url>');
  process.exit(2);
}

const OUTPUT_DIR = path.resolve('screenshots');
const CONCURRENCY = Math.max(1, Number(process.env.CONCURRENCY || 3));
const NAVIGATION_TIMEOUT_MS = Math.max(1000, Number(process.env.NAVIGATION_TIMEOUT_MS || 30000));
const FULL_PAGE = process.env.FULL_PAGE === '1';
const ALLOW_EXTERNAL = process.env.ALLOW_EXTERNAL === '1';
const USER_AGENT = process.env.USER_AGENT || 'SitemapScreenshotBot/1.0';

const parser = new XMLParser({
  ignoreAttributes: false,
  attributeNamePrefix: '@_',
  removeNSPrefix: true,
  parseTagValue: false,
  trimValues: true,
});

function asArray(value) {
  if (value === undefined || value === null) return [];
  return Array.isArray(value) ? value : [value];
}

function locText(entry) {
  const loc = typeof entry === 'string' ? entry : entry?.loc;
  return typeof loc === 'string' ? loc.trim() : '';
}

async function readUrl(url) {
  const response = await fetch(url, { redirect: 'follow' });
  if (!response.ok) throw new Error(`HTTP ${response.status} fetching ${url}`);
  let bytes = Buffer.from(await response.arrayBuffer());
  if (url.toLowerCase().split('?')[0].endsWith('.gz')) bytes = gunzipSync(bytes);
  return bytes.toString('utf8');
}

function isRobotsUrl(url) {
  return new URL(url).pathname.endsWith('/robots.txt');
}

async function sitemapLocationsFromRobots(robotsUrl) {
  const body = await readUrl(robotsUrl);
  const base = new URL(robotsUrl);
  return body.split(/\r?\n/)
    .map(line => line.match(/^\s*sitemap\s*:\s*(\S+)\s*$/i)?.[1])
    .filter(Boolean)
    .map(value => new URL(value, base).href);
}

async function parseSitemap(sitemapUrl, seenSitemaps, pageUrls) {
  const normalized = new URL(sitemapUrl).href;
  if (seenSitemaps.has(normalized)) return;
  seenSitemaps.add(normalized);

  const xml = await readUrl(normalized);
  const document = parser.parse(xml);
  if (document.sitemapindex) {
    const children = asArray(document.sitemapindex.sitemap)
      .map(locText).filter(Boolean);
    for (const child of children) {
      await parseSitemap(new URL(child, normalized).href, seenSitemaps, pageUrls);
    }
    return;
  }
  if (document.urlset) {
    for (const entry of asArray(document.urlset.url)) {
      const loc = locText(entry);
      if (loc) pageUrls.add(new URL(loc, normalized).href);
    }
    return;
  }
  throw new Error(`Unrecognized sitemap XML at ${normalized}`);
}

function safeTarget(url) {
  const parsed = new URL(url);
  const digest = createHash('sha256').update(url).digest('hex').slice(0, 16);
  const pathPart = decodeURIComponent(parsed.pathname)
    .split('/')
    .filter(Boolean)
    .map(part => part.replace(/[^a-zA-Z0-9._-]/g, '_'))
    .filter(Boolean)
    .slice(-4)
    .join('__') || 'home';
  return path.join(OUTPUT_DIR, `${parsed.hostname}__${pathPart}__${digest}.png`);
}

async function mapWithConcurrency(items, limit, task) {
  let next = 0;
  const workers = Array.from({ length: Math.min(limit, items.length) }, async () => {
    while (true) {
      const index = next++;
      if (index >= items.length) return;
      await task(items[index]);
    }
  });
  await Promise.all(workers);
}

const initial = new URL(startUrl);
let sitemapStarts;
if (isRobotsUrl(initial.href)) {
  sitemapStarts = await sitemapLocationsFromRobots(initial.href);
  if (sitemapStarts.length === 0) throw new Error(`No Sitemap: entries found in ${initial.href}`);
} else {
  sitemapStarts = [initial.href];
}

const seenSitemaps = new Set();
const pageUrls = new Set();
for (const sitemap of sitemapStarts) {
  await parseSitemap(sitemap, seenSitemaps, pageUrls);
}

const seedHosts = new Set(sitemapStarts.map(url => new URL(url).hostname));
const urls = [...pageUrls].filter(url => {
  try {
    const parsed = new URL(url);
    return ['http:', 'https:'].includes(parsed.protocol) &&
      (ALLOW_EXTERNAL || seedHosts.has(parsed.hostname));
  } catch {
    return false;
  }
}).sort();

await mkdir(OUTPUT_DIR, { recursive: true });
const manifest = createWriteStream(path.join(OUTPUT_DIR, 'results.jsonl'), { flags: 'w' });
const browser = await chromium.launch({ headless: true });
const context = await browser.newContext({
  viewport: { width: 1440, height: 1000 },
  deviceScaleFactor: 1,
  userAgent: USER_AGENT,
});
let failed = 0;

try {
  await mapWithConcurrency(urls, CONCURRENCY, async url => {
    const page = await context.newPage();
    const outputPath = safeTarget(url);
    let result;
    try {
      const response = await page.goto(url, {
        waitUntil: 'domcontentloaded',
        timeout: NAVIGATION_TIMEOUT_MS,
      });
      // A non-2xx response can still render a useful page; record its status.
      await page.screenshot({ path: outputPath, fullPage: FULL_PAGE });
      result = {
        url,
        ok: true,
        status: response?.status() ?? null,
        screenshot: path.basename(outputPath),
      };
      console.log(`OK ${response?.status() ?? 'no response'} ${url}`);
    } catch (error) {
      failed++;
      result = { url, ok: false, error: String(error?.message || error) };
      console.error(`FAIL ${url}: ${result.error}`);
    } finally {
      await page.close();
      if (!manifest.write(`${JSON.stringify(result)}\n`)) await once(manifest, 'drain');
    }
  });
} finally {
  await context.close();
  await browser.close();
  manifest.end();
  await once(manifest, 'finish');
}

console.log(`Finished: ${urls.length} URLs, ${failed} failures. Results: ${path.join(OUTPUT_DIR, 'results.jsonl')}`);
if (failed) process.exitCode = 1;

Run it with either a sitemap URL or a robots.txt URL containing Sitemap: records:

node screenshot-sitemap.mjs https://example.com/sitemap.xml
node screenshot-sitemap.mjs https://example.com/robots.txt

The output directory contains one PNG per accepted URL and results.jsonl, with a success or failure record for each URL. A URL hash is included in each filename so paths with the same ending do not overwrite each other. The script filters page URLs to the hostnames of the starting sitemap files by default. Set ALLOW_EXTERNAL=1 only if capturing external hosts listed in the sitemap is intentional.

3. Understand and adjust the capture settings

Setting Default How to change it
Viewport dimensions 1440 × 1000 CSS pixels Change viewport in browser.newContext().
Scale 1 device pixel per CSS pixel Set deviceScaleFactor, for example to 2 for higher-density output.
Capture mode Viewport screenshot Run with FULL_PAGE=1 for the full scrollable document.
Concurrency 3 pages at a time Set CONCURRENCY=2 or another positive value.
Navigation timeout 30 seconds Set NAVIGATION_TIMEOUT_MS=60000 to allow slower pages more time.
User agent SitemapScreenshotBot/1.0 Set USER_AGENT to an appropriate, truthful identifier.
Scope Only sitemap seed hostnames ALLOW_EXTERNAL=1 includes external HTTP(S) page URLs.

Use fullPage: true when below-the-fold content matters. It can make very tall images and does not guarantee that every lazy-loaded or dynamic element has appeared. For pages that load content after scrolling, add an intentional scroll-and-wait routine or use a selector wait for the content you need. Viewport captures are usually easier to compare consistently across a large URL set.

Playwright also supports screenshot output format and quality, clipping, and styles; see the Page screenshot API. JPEG and WebP settings can reduce storage compared with PNG, with format-specific quality tradeoffs. Choose a format deliberately if you change the script: extensions should match the selected format. For repeatable visual comparisons, keep browser version, operating system, viewport, scale, fonts, and capture settings stable. Playwright notes that rendering can vary with host OS, browser version, settings, hardware, power source, and headless mode in its visual comparison guidance.

4. Sitemap and crawl edge cases

Sitemap indexes and duplicates

The script recognizes both <urlset> and <sitemapindex>, follows index entries, and tracks sitemap locations and page URLs in sets. This handles repeated locations and nested index references without capturing the same page URL twice. The Sitemap Protocol sets limits of 50,000 URLs and 50 MB per sitemap file, and 50,000 sitemap entries and 50 MB per index. Treat these as format limits, not a promise that a large capture run will be quick or small.

Compressed sitemap files

The script decompresses a sitemap whose URL ends in .gz. If a server delivers a compressed file at a URL without that suffix, adapt readUrl() to inspect the response headers or use a known sitemap format. A decompression or XML parse error is surfaced as a script error rather than silently producing an incomplete list.

Scope, redirects, and unusual URLs

Page URLs are resolved as absolute URLs, and only HTTP and HTTPS are accepted. Host filtering uses the sitemap seed hostnames, which helps avoid accidentally capturing unrelated domains. If you use a CDN or intentionally list subdomains, inspect the resulting scope and change the check deliberately. Sitemap locations themselves may point to other hosts so indexes can be expanded; validate those locations if the sitemap source is not trusted. Redirects during fetching and page navigation are followed by the underlying APIs. The JSONL record retains the listed URL and final response status is recorded when available.

Robots.txt is discovery and guidance

A sitemap can be discovered from a Sitemap: line in robots.txt. The script supports that discovery route, but it does not implement robots.txt allow/disallow matching. Inspect and honor the applicable crawl rules before running captures, keep request concurrency modest, and capture only sites and pages you are authorized to access. RFC 9309 explains that robots.txt is crawler guidance, not access authorization: it does not grant access to protected material. See RFC 9309.

5. Authentication, waits, and browser context

For pages that require a legitimate authenticated session, provide only credentials you are authorized to use. For a basic header, add extraHTTPHeaders when creating the context:

const context = await browser.newContext({
  viewport: { width: 1440, height: 1000 },
  extraHTTPHeaders: { Authorization: `Bearer ${process.env.SITE_TOKEN}` },
});

Do not put secrets in source control or print them in logs. For cookie-based sessions, use await context.addCookies([...]) with the site’s required cookie fields before creating pages, or load a storage state created through an authorized sign-in flow. Avoid reusing a production user’s session for broad automated jobs unless that use is approved.

domcontentloaded is a practical default for broad runs because pages with analytics or long-running connections may never reach network idle. If a specific element indicates that the page is ready, wait for it before taking the screenshot:

await page.goto(url, { waitUntil: 'domcontentloaded', timeout: NAVIGATION_TIMEOUT_MS });
await page.locator('main').waitFor({ state: 'visible', timeout: 10000 });
await page.screenshot({ path: outputPath, fullPage: FULL_PAGE });

Replace main with a selector that exists on the target pages. A fixed delay is possible with await page.waitForTimeout(1000), but a meaningful selector is generally less wasteful and more reliable. For a heterogeneous sitemap, a selector wait can itself fail on pages with different layouts; handle that as a per-URL failure or define page-specific readiness rules.

6. Run large jobs reliably

  • Keep concurrency bounded. More pages in parallel use more memory and create more load on the target site. Start small and increase only when the site and machine can handle it.
  • Persist progress. The JSONL manifest is written as each page finishes, so completed results remain available if a later navigation fails or the process stops. Rerunning the script currently recaptures the list; use the manifest to build a resume filter if needed.
  • Separate collection from capture for very large inputs. Save the deduplicated URL list and process it in batches to control runtime, storage, and restart costs.
  • Track more than HTTP status. A navigation may return a 404 or 500 and still render a page. This script records the status but captures it; decide whether such pages belong in your report.
  • Set an explicit timeout. A timeout bounds the wait per navigation, but does not guarantee a fixed total runtime because pages vary in speed and the run may retry nothing.
  • Keep the environment stable for comparisons. Pin the browser/runtime environment used by your job and keep viewport and rendering settings the same between runs.

Storage depends on page length, image format, viewport, and content; this workflow has no fixed per-capture cost, but consumes your own compute time, memory, network capacity, and disk space. Estimate storage from a representative sample and monitor free space before a full run. The script intentionally does not retry automatically: retries can help transient failures, but they also add load and can conceal persistent access or page errors unless attempts are recorded separately.

7. Troubleshooting

Symptom Likely cause Fix
Executable doesn't exist or browser launch fails Chromium was not installed for Playwright. Run npx playwright install chromium in the project environment.
Usage: node screenshot-sitemap.mjs ... No starting URL was passed. Provide a sitemap or robots.txt URL as the first command-line argument.
Unrecognized sitemap XML The URL returned HTML, an error page, or a sitemap shape the script does not handle. Open the URL, confirm it returns sitemap XML, and check whether the format uses a namespace or structure that needs parser handling.
HTTP 403 or HTTP 404 while fetching sitemap The sitemap is unavailable to this request, the URL is wrong, or access is restricted. Confirm the exact sitemap URL and permitted access. Add appropriate authorized fetch headers if required.
Compressed sitemap parse or gunzip error The URL ends in .gz but the body is not gzip data, or the download is truncated. Check the response and URL, then adjust decompression handling for the server’s actual response.
Pages missing from the output The sitemap omits them, a referenced sitemap failed, a URL was filtered by host scope, or the page failed. Compare sitemap entries with results.jsonl, inspect all index files, and review the host filter and failed records.
Blank, incomplete, or loading screenshot The page needs more time, a readiness condition, authentication, or client-side content to render. Use an appropriate selector wait, authorized session state, or a longer timeout. Keep per-page failures visible.
Images or lower sections are absent The default capture is viewport-only, or content is lazy-loaded. Set FULL_PAGE=1; for lazy content, add deliberate scrolling and wait for the needed elements before capture.
Files appear overwritten or hard to identify Inspecting only the URL suffix can be ambiguous. The provided names include hostname, path fragments, and a URL hash. Keep the manifest with the files to map them back to listed URLs.
Out of memory or very slow run Concurrency or full-page image dimensions are too large for available resources. Lower CONCURRENCY, use viewport screenshots, or split the URL list into batches.

8. Or skip the browser setup

If you need screenshots but do not want to install and run a browser for each job, ScreenshotNeo provides a screenshot API. Its one-call endpoint captures one URL per request; for a sitemap, your script can still enumerate the URLs and call the API for each one. See the ScreenshotNeo API docs.

curl -G "https://api.screenshotneo.com/v1/shot" -d access_key=YOUR_API_KEY --data-urlencode url=https://example.com -o shot.webp
import requests

r = requests.get(
    "https://api.screenshotneo.com/v1/shot",
    params={"access_key": "YOUR_API_KEY", "url": "https://example.com"},
    timeout=90,
)
open("shot.webp", "wb").write(r.content)
const q = new URLSearchParams({ access_key: 'YOUR_API_KEY', url: 'https://example.com' });
const res = await fetch(`https://api.screenshotneo.com/v1/shot?${q}`);
if (!res.ok) throw new Error(`Screenshot request failed: ${res.status}`);
await Bun.write('shot.webp', res);

The Node.js snippet uses Bun’s file-writing helper for brevity. In Node.js, save the response with const fs = await import('node:fs/promises'); await fs.writeFile('shot.webp', Buffer.from(await res.arrayBuffer()));. ScreenshotNeo removes cookie banners, popups, and chat widgets before the shot; bot checks, blank pages, and failed loads are never billed; its MCP server lets AI agents take screenshots; and 1,000 screenshots a month are free with no card, with paid plans starting at $5 for 3,000. Sign up for 1,000 free screenshots a month, with no card required.

FAQ

Does a sitemap guarantee that every page on the site is captured?

No. The script captures accepted URLs represented in the sitemap files it successfully processes. A sitemap may not list every reachable URL.

Should I use a sitemap URL or robots.txt?

Use the sitemap URL when you know it. Use robots.txt when you want to discover sitemap locations published there; the script supports Sitemap: records.

Does the script enforce robots.txt rules?

No. Sitemap discovery is separate from crawl-rule evaluation. Inspect and honor applicable rules before running the capture.

Why can screenshots differ between runs?

Page content can change, and browser rendering can vary with the operating system, browser version, settings, hardware, and headless mode. Keep the environment and capture settings stable when comparing results.

Can I capture PDFs with this script?

This Playwright example saves screenshots. Playwright also has PDF capabilities in supported browser contexts; use the official Page PDF API documentation for its options and limitations.