/** * Internal link checker for the built site. * * Walks every .html file in dist/, pulls out every href and src that stays on this * origin, and resolves it against what the build actually emitted. Three failures are * reported, all of them things a visitor would hit: * * missing the target route was never emitted (a 404 on a real link) * anchor the route exists but the #id it points at is not in that page * asset a referenced local file (css, js, image) is not in dist/ * * Query strings are stripped before resolving: /mints?sort=rating is /mints. Links * that leave the origin (http(s)://other, mailto:, wss:) are counted and skipped. * * Usage: node scripts/check-links.mjs [distDir] * Exits non-zero when anything is broken, so it can gate a deploy. */ import { readdir, readFile, stat } from 'node:fs/promises'; import { existsSync } from 'node:fs'; import path from 'node:path'; const dist = path.resolve(process.argv[2] ?? 'dist'); if (!existsSync(dist)) { console.error(`No such directory: ${dist}. Run \`pnpm build\` first.`); process.exit(2); } /** Every file under dir, as paths relative to dir with forward slashes. */ async function walk(dir, base = dir) { const out = []; for (const entry of await readdir(dir, { withFileTypes: true })) { const full = path.join(dir, entry.name); if (entry.isDirectory()) out.push(...(await walk(full, base))); else out.push(path.relative(base, full).split(path.sep).join('/')); } return out; } const files = await walk(dist); const fileSet = new Set(files); const pages = files.filter((f) => f.endsWith('.html')); /** * Routes a visitor can reach. Astro writes /mints as mints/index.html, and the site * runs with trailingSlash: 'ignore', so both spellings resolve. */ /* * `/404` counts as a route. * * It used to be skipped, because nothing linked to it and it exists to be served by * `error_page` rather than navigated to. The language switcher changed that: on a 404, * every language's version of the 404 is a real destination, and the nginx block in the * README resolves `/404` through its `$uri.html` rule exactly as it resolves the * prefixed ones through their directory index. */ const routes = new Set(); for (const file of pages) { const route = file === 'index.html' ? '/' : '/' + file.replace(/\/index\.html$/, '').replace(/\.html$/, ''); routes.add(route); if (route !== '/') routes.add(route + '/'); } /** #ids present in each page, so anchor links can be checked against the real markup. */ const idsByRoute = new Map(); const HREF = /(?:href|src)\s*=\s*"([^"]*)"/gi; const ASSET_EXT = /\.(css|js|mjs|json|xml|txt|png|jpe?g|gif|svg|webp|avif|ico|woff2?|ttf|otf|map|webmanifest|pdf|mp4|webm)$/i; const ID = /\sid\s*=\s*"([^"]+)"/gi; const problems = []; const seen = new Map(); // target -> Set of pages linking to it let external = 0; for (const file of pages) { const html = await readFile(path.join(dist, file), 'utf8'); const route = file === 'index.html' ? '/' : '/' + file.replace(/\/index\.html$/, '').replace(/\.html$/, ''); const ids = new Set(); for (const m of html.matchAll(ID)) ids.add(m[1]); idsByRoute.set(route, ids); idsByRoute.set(route + '/', ids); for (const m of html.matchAll(HREF)) { const raw = m[1].trim(); if (!raw || raw.startsWith('#') || raw.startsWith('data:')) continue; if (/^[a-z][a-z0-9+.-]*:/i.test(raw) || raw.startsWith('//')) { external++; continue; } if (!raw.startsWith('/')) continue; // relative asset paths: Astro does not emit any const [pathPart, hashPart] = raw.split('#'); const target = (pathPart ?? '').split('?')[0] || '/'; const key = hashPart ? `${target}#${hashPart}` : target; if (!seen.has(key)) seen.set(key, new Set()); seen.get(key).add(route); // The API serves /api/* and /icons/* in production; nothing under them is in dist. if (target.startsWith('/api/') || target.startsWith('/icons/')) continue; const asFile = target.replace(/^\//, ''); // Extension based, from a fixed list: mint slugs are domains, so "ends in a dot // and some letters" would call /mint/cashu.boats an asset and look for the file. const isAsset = ASSET_EXT.test(asFile); if (isAsset) { if (!fileSet.has(asFile)) problems.push({ kind: 'asset', target, from: route }); continue; } if (!routes.has(target)) { problems.push({ kind: 'missing', target, from: route }); continue; } if (hashPart) { const known = idsByRoute.get(target); // The target page may not be parsed yet: anchors are re-checked in a second pass. if (known && !known.has(hashPart)) problems.push({ kind: 'anchor', target: `${target}#${hashPart}`, from: route }); } } } // Second pass for anchors whose target page had not been read yet on the first pass. for (const [key, froms] of seen) { if (!key.includes('#')) continue; const [target, hash] = key.split('#'); if (!routes.has(target)) continue; const ids = idsByRoute.get(target); if (!ids || ids.has(hash)) continue; for (const from of froms) { if (!problems.some((p) => p.kind === 'anchor' && p.target === key && p.from === from)) { problems.push({ kind: 'anchor', target: key, from }); } } } const internalTargets = [...seen.keys()].sort(); console.log(`Checked ${pages.length} pages, ${routes.size / 1} route spellings, ${internalTargets.length} distinct internal targets (${external} external links skipped).`); if (process.env['LIST_TARGETS']) { for (const target of internalTargets) { const froms = [...seen.get(target)].sort(); const shown = froms.length > 4 ? `${froms.slice(0, 4).join(', ')} +${froms.length - 4} more` : froms.join(', '); console.log(` ${target.padEnd(34)} <- ${shown}`); } } if (problems.length === 0) { console.log('No broken internal links.'); process.exit(0); } const label = { missing: 'route not emitted', anchor: 'no such id on the target page', asset: 'file not in dist' }; console.log(`\n${problems.length} problem(s):`); for (const p of problems) console.log(` [${label[p.kind]}] ${p.target} linked from ${p.from}`); process.exit(1);