Files
CashuMints.space/web/scripts/check-links.mjs
T
2026-08-20 22:41:25 +02:00

155 lines
6.0 KiB
JavaScript

/**
* Internal link checker for the built site.
*
* Walks every .html file in dist/, pulls out every href and src that stays on this
* origin, and resolves it against what the build actually emitted. Three failures are
* reported, all of them things a visitor would hit:
*
* missing the target route was never emitted (a 404 on a real link)
* anchor the route exists but the #id it points at is not in that page
* asset a referenced local file (css, js, image) is not in dist/
*
* Query strings are stripped before resolving: /mints?sort=rating is /mints. Links
* that leave the origin (http(s)://other, mailto:, wss:) are counted and skipped.
*
* Usage: node scripts/check-links.mjs [distDir]
* Exits non-zero when anything is broken, so it can gate a deploy.
*/
import { readdir, readFile, stat } from 'node:fs/promises';
import { existsSync } from 'node:fs';
import path from 'node:path';
const dist = path.resolve(process.argv[2] ?? 'dist');
if (!existsSync(dist)) {
console.error(`No such directory: ${dist}. Run \`pnpm build\` first.`);
process.exit(2);
}
/** Every file under dir, as paths relative to dir with forward slashes. */
async function walk(dir, base = dir) {
const out = [];
for (const entry of await readdir(dir, { withFileTypes: true })) {
const full = path.join(dir, entry.name);
if (entry.isDirectory()) out.push(...(await walk(full, base)));
else out.push(path.relative(base, full).split(path.sep).join('/'));
}
return out;
}
const files = await walk(dist);
const fileSet = new Set(files);
const pages = files.filter((f) => f.endsWith('.html'));
/**
* Routes a visitor can reach. Astro writes /mints as mints/index.html, and the site
* runs with trailingSlash: 'ignore', so both spellings resolve.
*/
/*
* `/404` counts as a route.
*
* It used to be skipped, because nothing linked to it and it exists to be served by
* `error_page` rather than navigated to. The language switcher changed that: on a 404,
* every language's version of the 404 is a real destination, and the nginx block in the
* README resolves `/404` through its `$uri.html` rule exactly as it resolves the
* prefixed ones through their directory index.
*/
const routes = new Set();
for (const file of pages) {
const route = file === 'index.html' ? '/' : '/' + file.replace(/\/index\.html$/, '').replace(/\.html$/, '');
routes.add(route);
if (route !== '/') routes.add(route + '/');
}
/** #ids present in each page, so anchor links can be checked against the real markup. */
const idsByRoute = new Map();
const HREF = /(?:href|src)\s*=\s*"([^"]*)"/gi;
const ASSET_EXT = /\.(css|js|mjs|json|xml|txt|png|jpe?g|gif|svg|webp|avif|ico|woff2?|ttf|otf|map|webmanifest|pdf|mp4|webm)$/i;
const ID = /\sid\s*=\s*"([^"]+)"/gi;
const problems = [];
const seen = new Map(); // target -> Set of pages linking to it
let external = 0;
for (const file of pages) {
const html = await readFile(path.join(dist, file), 'utf8');
const route = file === 'index.html' ? '/' : '/' + file.replace(/\/index\.html$/, '').replace(/\.html$/, '');
const ids = new Set();
for (const m of html.matchAll(ID)) ids.add(m[1]);
idsByRoute.set(route, ids);
idsByRoute.set(route + '/', ids);
for (const m of html.matchAll(HREF)) {
const raw = m[1].trim();
if (!raw || raw.startsWith('#') || raw.startsWith('data:')) continue;
if (/^[a-z][a-z0-9+.-]*:/i.test(raw) || raw.startsWith('//')) { external++; continue; }
if (!raw.startsWith('/')) continue; // relative asset paths: Astro does not emit any
const [pathPart, hashPart] = raw.split('#');
const target = (pathPart ?? '').split('?')[0] || '/';
const key = hashPart ? `${target}#${hashPart}` : target;
if (!seen.has(key)) seen.set(key, new Set());
seen.get(key).add(route);
// The API serves /api/* and /icons/* in production; nothing under them is in dist.
if (target.startsWith('/api/') || target.startsWith('/icons/')) continue;
const asFile = target.replace(/^\//, '');
// Extension based, from a fixed list: mint slugs are domains, so "ends in a dot
// and some letters" would call /mint/cashu.boats an asset and look for the file.
const isAsset = ASSET_EXT.test(asFile);
if (isAsset) {
if (!fileSet.has(asFile)) problems.push({ kind: 'asset', target, from: route });
continue;
}
if (!routes.has(target)) {
problems.push({ kind: 'missing', target, from: route });
continue;
}
if (hashPart) {
const known = idsByRoute.get(target);
// The target page may not be parsed yet: anchors are re-checked in a second pass.
if (known && !known.has(hashPart)) problems.push({ kind: 'anchor', target: `${target}#${hashPart}`, from: route });
}
}
}
// Second pass for anchors whose target page had not been read yet on the first pass.
for (const [key, froms] of seen) {
if (!key.includes('#')) continue;
const [target, hash] = key.split('#');
if (!routes.has(target)) continue;
const ids = idsByRoute.get(target);
if (!ids || ids.has(hash)) continue;
for (const from of froms) {
if (!problems.some((p) => p.kind === 'anchor' && p.target === key && p.from === from)) {
problems.push({ kind: 'anchor', target: key, from });
}
}
}
const internalTargets = [...seen.keys()].sort();
console.log(`Checked ${pages.length} pages, ${routes.size / 1} route spellings, ${internalTargets.length} distinct internal targets (${external} external links skipped).`);
if (process.env['LIST_TARGETS']) {
for (const target of internalTargets) {
const froms = [...seen.get(target)].sort();
const shown = froms.length > 4 ? `${froms.slice(0, 4).join(', ')} +${froms.length - 4} more` : froms.join(', ');
console.log(` ${target.padEnd(34)} <- ${shown}`);
}
}
if (problems.length === 0) {
console.log('No broken internal links.');
process.exit(0);
}
const label = { missing: 'route not emitted', anchor: 'no such id on the target page', asset: 'file not in dist' };
console.log(`\n${problems.length} problem(s):`);
for (const p of problems) console.log(` [${label[p.kind]}] ${p.target} linked from ${p.from}`);
process.exit(1);