/** * The i18n SEO check, over the built site. * * `check-links.mjs` answers "does every link go somewhere". This answers the question * that matters once a site is in three languages: does every page correctly declare * which language it is, which page it is, and where its siblings are. Those three * declarations are what stop three translations of one page being indexed as three * duplicates, and every one of them is easy to get subtly wrong in a way nothing else * catches. * * Seven checks per page: * * lang matches the locale its URL is under * canonical present, absolute, and pointing at itself (never at another locale) * alternates one per locale, plus exactly one x-default * absolute every alternate href is absolute * targets every alternate href is a page the build actually emitted * reciprocal every alternate lists this page back, under this page's own hreflang * x-default points at the default locale's version of this same page * * Plus, over the sitemap: every emitted page appears exactly once, with its own full * xhtml:link alternate set. * * Usage: node scripts/check-hreflang.mjs [distDir] * Exits non-zero on any problem, so it can gate a deploy. */ import { readdir, readFile } from 'node:fs/promises'; import { existsSync } from 'node:fs'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); const dist = path.resolve(process.argv[2] ?? path.join(root, 'dist')); if (!existsSync(dist)) { console.error(`No such directory: ${dist}. Run \`pnpm build\` first.`); process.exit(2); } const { DEFAULT_LOCALE, LOCALE_CODES } = await import(path.join(root, 'src/i18n/locales.mjs')); const PREFIXED = LOCALE_CODES.filter((code) => code !== DEFAULT_LOCALE); /* ---------- what the build emitted ---------- */ async function walk(dir, base = dir) { const out = []; for (const entry of await readdir(dir, { withFileTypes: true })) { const full = path.join(dir, entry.name); if (entry.isDirectory()) out.push(...(await walk(full, base))); else out.push(path.relative(base, full).split(path.sep).join('/')); } return out; } const files = await walk(dist); const pages = files.filter((f) => f.endsWith('.html')); /** dist-relative file to the route a visitor asks for. */ const routeOf = (file) => file === 'index.html' ? '/' : '/' + file.replace(/\/index\.html$/, '').replace(/\.html$/, ''); /** The locale a route is under, and the route with that prefix removed. */ function split(route) { const first = route.split('/').filter(Boolean)[0]; if (first && PREFIXED.includes(first)) { const rest = route.slice(first.length + 1); return { locale: first, path: rest === '' ? '/' : rest }; } return { locale: DEFAULT_LOCALE, path: route }; } const problems = []; const fail = (page, kind, detail) => problems.push({ page, kind, detail }); /* ---------- read every page ---------- */ const LANG = /]*\slang="([^"]*)"/i; const CANONICAL = /]*rel="canonical"[^>]*href="([^"]*)"/i; const ROBOTS = /]*name="robots"[^>]*content="([^"]*)"/i; const ALTERNATE = /]*rel="alternate"[^>]*hreflang="([^"]*)"[^>]*href="([^"]*)"/gi; const seen = new Map(); // route -> { locale, path, lang, canonical, alternates: Map } const routes = new Set(); for (const file of pages) { const route = routeOf(file); routes.add(route); const html = await readFile(path.join(dist, file), 'utf8'); const alternates = new Map(); for (const match of html.matchAll(ALTERNATE)) alternates.set(match[1], match[2]); seen.set(route, { ...split(route), route, lang: LANG.exec(html)?.[1] ?? null, canonical: CANONICAL.exec(html)?.[1] ?? null, robots: ROBOTS.exec(html)?.[1] ?? null, alternates, }); } /* * The 404 page is the one route with no place in the alternate graph. * * It is not a page anyone links to or a crawler should index; it is what a web server * hands back for a URL that does not exist. It still gets a `lang` and it still gets * one per locale, which is what the checks below verify, but it is deliberately absent * from the sitemap and it is not required to be reciprocal with anything. */ const isOffGraph = (page) => page.path === '/404'; /** * A page that exists but asks not to be indexed. * * Mint pages the site has never reached and nobody has reviewed carry this (see * `src/lib/seo.ts`). Unlike the 404 they are real pages: they keep their canonical and * their full alternate set, and only the sitemap leaves them out. */ const isNoindex = (page) => Boolean(page.robots?.includes('noindex')); /* ---------- the seven checks ---------- */ const origins = new Set(); for (const page of seen.values()) { if (page.lang !== page.locale) { fail(page.route, 'lang', ` under /${page.locale}/`); } if (isOffGraph(page)) { // No canonical, no alternates, and a robots meta saying so. Checked rather than // skipped: "the 404 has no alternate set" is a claim worth verifying, because the // easy mistake is for it to quietly acquire one. if (page.canonical) fail(page.route, 'noindex', `404 has a canonical: ${page.canonical}`); if (page.alternates.size > 0) { fail(page.route, 'noindex', `404 declares ${page.alternates.size} alternate(s)`); } if (!page.robots?.includes('noindex')) fail(page.route, 'noindex', '404 is missing noindex'); continue; } if (!page.canonical) { fail(page.route, 'canonical', 'no '); continue; } if (!/^https?:\/\//.test(page.canonical)) { fail(page.route, 'canonical', `relative: ${page.canonical}`); continue; } const canonicalUrl = new URL(page.canonical); origins.add(canonicalUrl.origin); // Self-canonical. A locale canonicalising to another locale tells a crawler this // page is a duplicate that should not be indexed, which is the exact opposite of // what the alternate set beside it is claiming. if (canonicalUrl.pathname.replace(/\/$/, '') !== page.route.replace(/\/$/, '')) { fail(page.route, 'canonical', `points at ${canonicalUrl.pathname}, not at itself`); } // One per locale, plus x-default, and nothing else. const expected = [...LOCALE_CODES, 'x-default']; for (const hreflang of expected) { if (!page.alternates.has(hreflang)) { fail(page.route, 'alternates', `missing hreflang="${hreflang}"`); } } for (const hreflang of page.alternates.keys()) { if (!expected.includes(hreflang)) { fail(page.route, 'alternates', `unexpected hreflang="${hreflang}"`); } } for (const [hreflang, href] of page.alternates) { if (!/^https?:\/\//.test(href)) { fail(page.route, 'absolute', `hreflang="${hreflang}" is relative: ${href}`); continue; } const target = new URL(href).pathname.replace(/\/$/, '') || '/'; if (!routes.has(target)) { fail(page.route, 'targets', `hreflang="${hreflang}" points at ${target}, which was not emitted`); continue; } if (hreflang === 'x-default') { const expectedDefault = split(target); if (expectedDefault.locale !== DEFAULT_LOCALE || expectedDefault.path !== page.path) { fail(page.route, 'x-default', `points at ${target}, not at the ${DEFAULT_LOCALE} version of ${page.path}`); } continue; } // Reciprocity: the page this one points at has to point back, at this page, under // this page's own hreflang. A set that is not reciprocal is discarded whole, so a // one-sided link is worth less than no link. const other = seen.get(target); const back = other?.alternates.get(page.locale); if (!back) { fail(page.route, 'reciprocal', `${target} does not list hreflang="${page.locale}"`); continue; } const backPath = new URL(back).pathname.replace(/\/$/, '') || '/'; if (backPath !== (page.route.replace(/\/$/, '') || '/')) { fail(page.route, 'reciprocal', `${target} lists hreflang="${page.locale}" as ${backPath}`); } } } if (origins.size > 1) { fail('(site)', 'origin', `canonical URLs use more than one origin: ${[...origins].join(', ')}`); } /* ---------- the sitemap ---------- */ const sitemapPath = path.join(dist, 'sitemap.xml'); let sitemapCount = 0; if (!existsSync(sitemapPath)) { fail('(sitemap)', 'sitemap', 'sitemap.xml was not emitted'); } else { const xml = await readFile(sitemapPath, 'utf8'); const blocks = [...xml.matchAll(/([\s\S]*?)<\/url>/g)].map((m) => m[1]); sitemapCount = blocks.length; const listed = new Map(); for (const block of blocks) { const loc = /([^<]*)<\/loc>/.exec(block)?.[1]; if (!loc) { fail('(sitemap)', 'sitemap', 'a block has no '); continue; } const route = new URL(loc).pathname.replace(/\/$/, '') || '/'; if (listed.has(route)) fail(route, 'sitemap', 'listed more than once'); const alternates = new Set( [...block.matchAll(/]*hreflang="([^"]*)"/g)].map((m) => m[1]), ); listed.set(route, alternates); for (const hreflang of [...LOCALE_CODES, 'x-default']) { if (!alternates.has(hreflang)) { fail(route, 'sitemap', `entry is missing xhtml:link hreflang="${hreflang}"`); } } } /* * The invariant, both ways round: every indexable page is listed, and nothing that * asks not to be indexed is. * * Checking it here rather than trusting the two files to agree is the point. The * page's robots meta and the sitemap are produced by different code from one shared * predicate, and this reads the built output of both, so the day someone changes one * without the other the build says so. */ for (const page of seen.values()) { if (isNoindex(page)) { if (listed.has(page.route)) { fail(page.route, 'sitemap', 'says noindex but is listed in the sitemap'); } continue; } if (!listed.has(page.route)) fail(page.route, 'sitemap', 'emitted but not in the sitemap'); } for (const route of listed.keys()) { if (!routes.has(route)) fail(route, 'sitemap', 'listed but never emitted'); } } /* ---------- report ---------- */ const localeCounts = LOCALE_CODES.map( (code) => `${code}: ${[...seen.values()].filter((p) => p.locale === code).length}`, ).join(', '); console.log( `check-hreflang: ${pages.length} pages (${localeCounts}), ${sitemapCount} sitemap entries.`, ); if (problems.length === 0) { console.log('Every page declares its language, canonicalises to itself, and has a complete reciprocal alternate set.'); process.exit(0); } const grouped = new Map(); for (const problem of problems) { if (!grouped.has(problem.kind)) grouped.set(problem.kind, []); grouped.get(problem.kind).push(problem); } console.log(`\n${problems.length} problem(s):`); for (const [kind, list] of grouped) { console.log(`\n [${kind}] ${list.length}`); for (const problem of list.slice(0, 15)) console.log(` ${problem.page}: ${problem.detail}`); if (list.length > 15) console.log(` … and ${list.length - 15} more`); } process.exit(1);