@@ -0,0 +1,300 @@
|
||||
/**
|
||||
* The i18n SEO check, over the built site.
|
||||
*
|
||||
* `check-links.mjs` answers "does every link go somewhere". This answers the question
|
||||
* that matters once a site is in three languages: does every page correctly declare
|
||||
* which language it is, which page it is, and where its siblings are. Those three
|
||||
* declarations are what stop three translations of one page being indexed as three
|
||||
* duplicates, and every one of them is easy to get subtly wrong in a way nothing else
|
||||
* catches.
|
||||
*
|
||||
* Seven checks per page:
|
||||
*
|
||||
* lang <html lang> matches the locale its URL is under
|
||||
* canonical present, absolute, and pointing at itself (never at another locale)
|
||||
* alternates one per locale, plus exactly one x-default
|
||||
* absolute every alternate href is absolute
|
||||
* targets every alternate href is a page the build actually emitted
|
||||
* reciprocal every alternate lists this page back, under this page's own hreflang
|
||||
* x-default points at the default locale's version of this same page
|
||||
*
|
||||
* Plus, over the sitemap: every emitted page appears exactly once, with its own full
|
||||
* xhtml:link alternate set.
|
||||
*
|
||||
* Usage: node scripts/check-hreflang.mjs [distDir]
|
||||
* Exits non-zero on any problem, so it can gate a deploy.
|
||||
*/
|
||||
import { readdir, readFile } from 'node:fs/promises';
|
||||
import { existsSync } from 'node:fs';
|
||||
import path from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
||||
const dist = path.resolve(process.argv[2] ?? path.join(root, 'dist'));
|
||||
|
||||
if (!existsSync(dist)) {
|
||||
console.error(`No such directory: ${dist}. Run \`pnpm build\` first.`);
|
||||
process.exit(2);
|
||||
}
|
||||
|
||||
const { DEFAULT_LOCALE, LOCALE_CODES } = await import(path.join(root, 'src/i18n/locales.mjs'));
|
||||
const PREFIXED = LOCALE_CODES.filter((code) => code !== DEFAULT_LOCALE);
|
||||
|
||||
/* ---------- what the build emitted ---------- */
|
||||
|
||||
async function walk(dir, base = dir) {
|
||||
const out = [];
|
||||
for (const entry of await readdir(dir, { withFileTypes: true })) {
|
||||
const full = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) out.push(...(await walk(full, base)));
|
||||
else out.push(path.relative(base, full).split(path.sep).join('/'));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
const files = await walk(dist);
|
||||
const pages = files.filter((f) => f.endsWith('.html'));
|
||||
|
||||
/** dist-relative file to the route a visitor asks for. */
|
||||
const routeOf = (file) =>
|
||||
file === 'index.html' ? '/' : '/' + file.replace(/\/index\.html$/, '').replace(/\.html$/, '');
|
||||
|
||||
/** The locale a route is under, and the route with that prefix removed. */
|
||||
function split(route) {
|
||||
const first = route.split('/').filter(Boolean)[0];
|
||||
if (first && PREFIXED.includes(first)) {
|
||||
const rest = route.slice(first.length + 1);
|
||||
return { locale: first, path: rest === '' ? '/' : rest };
|
||||
}
|
||||
return { locale: DEFAULT_LOCALE, path: route };
|
||||
}
|
||||
|
||||
const problems = [];
|
||||
const fail = (page, kind, detail) => problems.push({ page, kind, detail });
|
||||
|
||||
/* ---------- read every page ---------- */
|
||||
|
||||
const LANG = /<html[^>]*\slang="([^"]*)"/i;
|
||||
const CANONICAL = /<link[^>]*rel="canonical"[^>]*href="([^"]*)"/i;
|
||||
const ROBOTS = /<meta[^>]*name="robots"[^>]*content="([^"]*)"/i;
|
||||
const ALTERNATE = /<link[^>]*rel="alternate"[^>]*hreflang="([^"]*)"[^>]*href="([^"]*)"/gi;
|
||||
|
||||
const seen = new Map(); // route -> { locale, path, lang, canonical, alternates: Map }
|
||||
const routes = new Set();
|
||||
|
||||
for (const file of pages) {
|
||||
const route = routeOf(file);
|
||||
routes.add(route);
|
||||
const html = await readFile(path.join(dist, file), 'utf8');
|
||||
|
||||
const alternates = new Map();
|
||||
for (const match of html.matchAll(ALTERNATE)) alternates.set(match[1], match[2]);
|
||||
|
||||
seen.set(route, {
|
||||
...split(route),
|
||||
route,
|
||||
lang: LANG.exec(html)?.[1] ?? null,
|
||||
canonical: CANONICAL.exec(html)?.[1] ?? null,
|
||||
robots: ROBOTS.exec(html)?.[1] ?? null,
|
||||
alternates,
|
||||
});
|
||||
}
|
||||
|
||||
/*
|
||||
* The 404 page is the one route with no place in the alternate graph.
|
||||
*
|
||||
* It is not a page anyone links to or a crawler should index; it is what a web server
|
||||
* hands back for a URL that does not exist. It still gets a `lang` and it still gets
|
||||
* one per locale, which is what the checks below verify, but it is deliberately absent
|
||||
* from the sitemap and it is not required to be reciprocal with anything.
|
||||
*/
|
||||
const isOffGraph = (page) => page.path === '/404';
|
||||
|
||||
/**
|
||||
* A page that exists but asks not to be indexed.
|
||||
*
|
||||
* Mint pages the site has never reached and nobody has reviewed carry this (see
|
||||
* `src/lib/seo.ts`). Unlike the 404 they are real pages: they keep their canonical and
|
||||
* their full alternate set, and only the sitemap leaves them out.
|
||||
*/
|
||||
const isNoindex = (page) => Boolean(page.robots?.includes('noindex'));
|
||||
|
||||
/* ---------- the seven checks ---------- */
|
||||
|
||||
const origins = new Set();
|
||||
|
||||
for (const page of seen.values()) {
|
||||
if (page.lang !== page.locale) {
|
||||
fail(page.route, 'lang', `<html lang="${page.lang}"> under /${page.locale}/`);
|
||||
}
|
||||
|
||||
if (isOffGraph(page)) {
|
||||
// No canonical, no alternates, and a robots meta saying so. Checked rather than
|
||||
// skipped: "the 404 has no alternate set" is a claim worth verifying, because the
|
||||
// easy mistake is for it to quietly acquire one.
|
||||
if (page.canonical) fail(page.route, 'noindex', `404 has a canonical: ${page.canonical}`);
|
||||
if (page.alternates.size > 0) {
|
||||
fail(page.route, 'noindex', `404 declares ${page.alternates.size} alternate(s)`);
|
||||
}
|
||||
if (!page.robots?.includes('noindex')) fail(page.route, 'noindex', '404 is missing noindex');
|
||||
continue;
|
||||
}
|
||||
|
||||
if (!page.canonical) {
|
||||
fail(page.route, 'canonical', 'no <link rel="canonical">');
|
||||
continue;
|
||||
}
|
||||
if (!/^https?:\/\//.test(page.canonical)) {
|
||||
fail(page.route, 'canonical', `relative: ${page.canonical}`);
|
||||
continue;
|
||||
}
|
||||
|
||||
const canonicalUrl = new URL(page.canonical);
|
||||
origins.add(canonicalUrl.origin);
|
||||
|
||||
|
||||
// Self-canonical. A locale canonicalising to another locale tells a crawler this
|
||||
// page is a duplicate that should not be indexed, which is the exact opposite of
|
||||
// what the alternate set beside it is claiming.
|
||||
if (canonicalUrl.pathname.replace(/\/$/, '') !== page.route.replace(/\/$/, '')) {
|
||||
fail(page.route, 'canonical', `points at ${canonicalUrl.pathname}, not at itself`);
|
||||
}
|
||||
|
||||
// One per locale, plus x-default, and nothing else.
|
||||
const expected = [...LOCALE_CODES, 'x-default'];
|
||||
for (const hreflang of expected) {
|
||||
if (!page.alternates.has(hreflang)) {
|
||||
fail(page.route, 'alternates', `missing hreflang="${hreflang}"`);
|
||||
}
|
||||
}
|
||||
for (const hreflang of page.alternates.keys()) {
|
||||
if (!expected.includes(hreflang)) {
|
||||
fail(page.route, 'alternates', `unexpected hreflang="${hreflang}"`);
|
||||
}
|
||||
}
|
||||
|
||||
for (const [hreflang, href] of page.alternates) {
|
||||
if (!/^https?:\/\//.test(href)) {
|
||||
fail(page.route, 'absolute', `hreflang="${hreflang}" is relative: ${href}`);
|
||||
continue;
|
||||
}
|
||||
|
||||
const target = new URL(href).pathname.replace(/\/$/, '') || '/';
|
||||
if (!routes.has(target)) {
|
||||
fail(page.route, 'targets', `hreflang="${hreflang}" points at ${target}, which was not emitted`);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (hreflang === 'x-default') {
|
||||
const expectedDefault = split(target);
|
||||
if (expectedDefault.locale !== DEFAULT_LOCALE || expectedDefault.path !== page.path) {
|
||||
fail(page.route, 'x-default', `points at ${target}, not at the ${DEFAULT_LOCALE} version of ${page.path}`);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// Reciprocity: the page this one points at has to point back, at this page, under
|
||||
// this page's own hreflang. A set that is not reciprocal is discarded whole, so a
|
||||
// one-sided link is worth less than no link.
|
||||
const other = seen.get(target);
|
||||
const back = other?.alternates.get(page.locale);
|
||||
if (!back) {
|
||||
fail(page.route, 'reciprocal', `${target} does not list hreflang="${page.locale}"`);
|
||||
continue;
|
||||
}
|
||||
const backPath = new URL(back).pathname.replace(/\/$/, '') || '/';
|
||||
if (backPath !== (page.route.replace(/\/$/, '') || '/')) {
|
||||
fail(page.route, 'reciprocal', `${target} lists hreflang="${page.locale}" as ${backPath}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (origins.size > 1) {
|
||||
fail('(site)', 'origin', `canonical URLs use more than one origin: ${[...origins].join(', ')}`);
|
||||
}
|
||||
|
||||
/* ---------- the sitemap ---------- */
|
||||
|
||||
const sitemapPath = path.join(dist, 'sitemap.xml');
|
||||
let sitemapCount = 0;
|
||||
|
||||
if (!existsSync(sitemapPath)) {
|
||||
fail('(sitemap)', 'sitemap', 'sitemap.xml was not emitted');
|
||||
} else {
|
||||
const xml = await readFile(sitemapPath, 'utf8');
|
||||
const blocks = [...xml.matchAll(/<url>([\s\S]*?)<\/url>/g)].map((m) => m[1]);
|
||||
sitemapCount = blocks.length;
|
||||
|
||||
const listed = new Map();
|
||||
for (const block of blocks) {
|
||||
const loc = /<loc>([^<]*)<\/loc>/.exec(block)?.[1];
|
||||
if (!loc) {
|
||||
fail('(sitemap)', 'sitemap', 'a <url> block has no <loc>');
|
||||
continue;
|
||||
}
|
||||
const route = new URL(loc).pathname.replace(/\/$/, '') || '/';
|
||||
if (listed.has(route)) fail(route, 'sitemap', 'listed more than once');
|
||||
|
||||
const alternates = new Set(
|
||||
[...block.matchAll(/<xhtml:link[^>]*hreflang="([^"]*)"/g)].map((m) => m[1]),
|
||||
);
|
||||
listed.set(route, alternates);
|
||||
|
||||
for (const hreflang of [...LOCALE_CODES, 'x-default']) {
|
||||
if (!alternates.has(hreflang)) {
|
||||
fail(route, 'sitemap', `entry is missing xhtml:link hreflang="${hreflang}"`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/*
|
||||
* The invariant, both ways round: every indexable page is listed, and nothing that
|
||||
* asks not to be indexed is.
|
||||
*
|
||||
* Checking it here rather than trusting the two files to agree is the point. The
|
||||
* page's robots meta and the sitemap are produced by different code from one shared
|
||||
* predicate, and this reads the built output of both, so the day someone changes one
|
||||
* without the other the build says so.
|
||||
*/
|
||||
for (const page of seen.values()) {
|
||||
if (isNoindex(page)) {
|
||||
if (listed.has(page.route)) {
|
||||
fail(page.route, 'sitemap', 'says noindex but is listed in the sitemap');
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (!listed.has(page.route)) fail(page.route, 'sitemap', 'emitted but not in the sitemap');
|
||||
}
|
||||
for (const route of listed.keys()) {
|
||||
if (!routes.has(route)) fail(route, 'sitemap', 'listed but never emitted');
|
||||
}
|
||||
}
|
||||
|
||||
/* ---------- report ---------- */
|
||||
|
||||
const localeCounts = LOCALE_CODES.map(
|
||||
(code) => `${code}: ${[...seen.values()].filter((p) => p.locale === code).length}`,
|
||||
).join(', ');
|
||||
|
||||
console.log(
|
||||
`check-hreflang: ${pages.length} pages (${localeCounts}), ${sitemapCount} sitemap entries.`,
|
||||
);
|
||||
|
||||
if (problems.length === 0) {
|
||||
console.log('Every page declares its language, canonicalises to itself, and has a complete reciprocal alternate set.');
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
const grouped = new Map();
|
||||
for (const problem of problems) {
|
||||
if (!grouped.has(problem.kind)) grouped.set(problem.kind, []);
|
||||
grouped.get(problem.kind).push(problem);
|
||||
}
|
||||
|
||||
console.log(`\n${problems.length} problem(s):`);
|
||||
for (const [kind, list] of grouped) {
|
||||
console.log(`\n [${kind}] ${list.length}`);
|
||||
for (const problem of list.slice(0, 15)) console.log(` ${problem.page}: ${problem.detail}`);
|
||||
if (list.length > 15) console.log(` … and ${list.length - 15} more`);
|
||||
}
|
||||
process.exit(1);
|
||||
Reference in New Issue
Block a user