301 lines
11 KiB
JavaScript
301 lines
11 KiB
JavaScript
/**
|
|
* The i18n SEO check, over the built site.
|
|
*
|
|
* `check-links.mjs` answers "does every link go somewhere". This answers the question
|
|
* that matters once a site is in three languages: does every page correctly declare
|
|
* which language it is, which page it is, and where its siblings are. Those three
|
|
* declarations are what stop three translations of one page being indexed as three
|
|
* duplicates, and every one of them is easy to get subtly wrong in a way nothing else
|
|
* catches.
|
|
*
|
|
* Seven checks per page:
|
|
*
|
|
* lang <html lang> matches the locale its URL is under
|
|
* canonical present, absolute, and pointing at itself (never at another locale)
|
|
* alternates one per locale, plus exactly one x-default
|
|
* absolute every alternate href is absolute
|
|
* targets every alternate href is a page the build actually emitted
|
|
* reciprocal every alternate lists this page back, under this page's own hreflang
|
|
* x-default points at the default locale's version of this same page
|
|
*
|
|
* Plus, over the sitemap: every emitted page appears exactly once, with its own full
|
|
* xhtml:link alternate set.
|
|
*
|
|
* Usage: node scripts/check-hreflang.mjs [distDir]
|
|
* Exits non-zero on any problem, so it can gate a deploy.
|
|
*/
|
|
import { readdir, readFile } from 'node:fs/promises';
|
|
import { existsSync } from 'node:fs';
|
|
import path from 'node:path';
|
|
import { fileURLToPath } from 'node:url';
|
|
|
|
const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
const dist = path.resolve(process.argv[2] ?? path.join(root, 'dist'));
|
|
|
|
if (!existsSync(dist)) {
|
|
console.error(`No such directory: ${dist}. Run \`pnpm build\` first.`);
|
|
process.exit(2);
|
|
}
|
|
|
|
const { DEFAULT_LOCALE, LOCALE_CODES } = await import(path.join(root, 'src/i18n/locales.mjs'));
|
|
const PREFIXED = LOCALE_CODES.filter((code) => code !== DEFAULT_LOCALE);
|
|
|
|
/* ---------- what the build emitted ---------- */
|
|
|
|
async function walk(dir, base = dir) {
|
|
const out = [];
|
|
for (const entry of await readdir(dir, { withFileTypes: true })) {
|
|
const full = path.join(dir, entry.name);
|
|
if (entry.isDirectory()) out.push(...(await walk(full, base)));
|
|
else out.push(path.relative(base, full).split(path.sep).join('/'));
|
|
}
|
|
return out;
|
|
}
|
|
|
|
const files = await walk(dist);
|
|
const pages = files.filter((f) => f.endsWith('.html'));
|
|
|
|
/** dist-relative file to the route a visitor asks for. */
|
|
const routeOf = (file) =>
|
|
file === 'index.html' ? '/' : '/' + file.replace(/\/index\.html$/, '').replace(/\.html$/, '');
|
|
|
|
/** The locale a route is under, and the route with that prefix removed. */
|
|
function split(route) {
|
|
const first = route.split('/').filter(Boolean)[0];
|
|
if (first && PREFIXED.includes(first)) {
|
|
const rest = route.slice(first.length + 1);
|
|
return { locale: first, path: rest === '' ? '/' : rest };
|
|
}
|
|
return { locale: DEFAULT_LOCALE, path: route };
|
|
}
|
|
|
|
const problems = [];
|
|
const fail = (page, kind, detail) => problems.push({ page, kind, detail });
|
|
|
|
/* ---------- read every page ---------- */
|
|
|
|
const LANG = /<html[^>]*\slang="([^"]*)"/i;
|
|
const CANONICAL = /<link[^>]*rel="canonical"[^>]*href="([^"]*)"/i;
|
|
const ROBOTS = /<meta[^>]*name="robots"[^>]*content="([^"]*)"/i;
|
|
const ALTERNATE = /<link[^>]*rel="alternate"[^>]*hreflang="([^"]*)"[^>]*href="([^"]*)"/gi;
|
|
|
|
const seen = new Map(); // route -> { locale, path, lang, canonical, alternates: Map }
|
|
const routes = new Set();
|
|
|
|
for (const file of pages) {
|
|
const route = routeOf(file);
|
|
routes.add(route);
|
|
const html = await readFile(path.join(dist, file), 'utf8');
|
|
|
|
const alternates = new Map();
|
|
for (const match of html.matchAll(ALTERNATE)) alternates.set(match[1], match[2]);
|
|
|
|
seen.set(route, {
|
|
...split(route),
|
|
route,
|
|
lang: LANG.exec(html)?.[1] ?? null,
|
|
canonical: CANONICAL.exec(html)?.[1] ?? null,
|
|
robots: ROBOTS.exec(html)?.[1] ?? null,
|
|
alternates,
|
|
});
|
|
}
|
|
|
|
/*
|
|
* The 404 page is the one route with no place in the alternate graph.
|
|
*
|
|
* It is not a page anyone links to or a crawler should index; it is what a web server
|
|
* hands back for a URL that does not exist. It still gets a `lang` and it still gets
|
|
* one per locale, which is what the checks below verify, but it is deliberately absent
|
|
* from the sitemap and it is not required to be reciprocal with anything.
|
|
*/
|
|
const isOffGraph = (page) => page.path === '/404';
|
|
|
|
/**
|
|
* A page that exists but asks not to be indexed.
|
|
*
|
|
* Mint pages the site has never reached and nobody has reviewed carry this (see
|
|
* `src/lib/seo.ts`). Unlike the 404 they are real pages: they keep their canonical and
|
|
* their full alternate set, and only the sitemap leaves them out.
|
|
*/
|
|
const isNoindex = (page) => Boolean(page.robots?.includes('noindex'));
|
|
|
|
/* ---------- the seven checks ---------- */
|
|
|
|
const origins = new Set();
|
|
|
|
for (const page of seen.values()) {
|
|
if (page.lang !== page.locale) {
|
|
fail(page.route, 'lang', `<html lang="${page.lang}"> under /${page.locale}/`);
|
|
}
|
|
|
|
if (isOffGraph(page)) {
|
|
// No canonical, no alternates, and a robots meta saying so. Checked rather than
|
|
// skipped: "the 404 has no alternate set" is a claim worth verifying, because the
|
|
// easy mistake is for it to quietly acquire one.
|
|
if (page.canonical) fail(page.route, 'noindex', `404 has a canonical: ${page.canonical}`);
|
|
if (page.alternates.size > 0) {
|
|
fail(page.route, 'noindex', `404 declares ${page.alternates.size} alternate(s)`);
|
|
}
|
|
if (!page.robots?.includes('noindex')) fail(page.route, 'noindex', '404 is missing noindex');
|
|
continue;
|
|
}
|
|
|
|
if (!page.canonical) {
|
|
fail(page.route, 'canonical', 'no <link rel="canonical">');
|
|
continue;
|
|
}
|
|
if (!/^https?:\/\//.test(page.canonical)) {
|
|
fail(page.route, 'canonical', `relative: ${page.canonical}`);
|
|
continue;
|
|
}
|
|
|
|
const canonicalUrl = new URL(page.canonical);
|
|
origins.add(canonicalUrl.origin);
|
|
|
|
|
|
// Self-canonical. A locale canonicalising to another locale tells a crawler this
|
|
// page is a duplicate that should not be indexed, which is the exact opposite of
|
|
// what the alternate set beside it is claiming.
|
|
if (canonicalUrl.pathname.replace(/\/$/, '') !== page.route.replace(/\/$/, '')) {
|
|
fail(page.route, 'canonical', `points at ${canonicalUrl.pathname}, not at itself`);
|
|
}
|
|
|
|
// One per locale, plus x-default, and nothing else.
|
|
const expected = [...LOCALE_CODES, 'x-default'];
|
|
for (const hreflang of expected) {
|
|
if (!page.alternates.has(hreflang)) {
|
|
fail(page.route, 'alternates', `missing hreflang="${hreflang}"`);
|
|
}
|
|
}
|
|
for (const hreflang of page.alternates.keys()) {
|
|
if (!expected.includes(hreflang)) {
|
|
fail(page.route, 'alternates', `unexpected hreflang="${hreflang}"`);
|
|
}
|
|
}
|
|
|
|
for (const [hreflang, href] of page.alternates) {
|
|
if (!/^https?:\/\//.test(href)) {
|
|
fail(page.route, 'absolute', `hreflang="${hreflang}" is relative: ${href}`);
|
|
continue;
|
|
}
|
|
|
|
const target = new URL(href).pathname.replace(/\/$/, '') || '/';
|
|
if (!routes.has(target)) {
|
|
fail(page.route, 'targets', `hreflang="${hreflang}" points at ${target}, which was not emitted`);
|
|
continue;
|
|
}
|
|
|
|
if (hreflang === 'x-default') {
|
|
const expectedDefault = split(target);
|
|
if (expectedDefault.locale !== DEFAULT_LOCALE || expectedDefault.path !== page.path) {
|
|
fail(page.route, 'x-default', `points at ${target}, not at the ${DEFAULT_LOCALE} version of ${page.path}`);
|
|
}
|
|
continue;
|
|
}
|
|
|
|
// Reciprocity: the page this one points at has to point back, at this page, under
|
|
// this page's own hreflang. A set that is not reciprocal is discarded whole, so a
|
|
// one-sided link is worth less than no link.
|
|
const other = seen.get(target);
|
|
const back = other?.alternates.get(page.locale);
|
|
if (!back) {
|
|
fail(page.route, 'reciprocal', `${target} does not list hreflang="${page.locale}"`);
|
|
continue;
|
|
}
|
|
const backPath = new URL(back).pathname.replace(/\/$/, '') || '/';
|
|
if (backPath !== (page.route.replace(/\/$/, '') || '/')) {
|
|
fail(page.route, 'reciprocal', `${target} lists hreflang="${page.locale}" as ${backPath}`);
|
|
}
|
|
}
|
|
}
|
|
|
|
if (origins.size > 1) {
|
|
fail('(site)', 'origin', `canonical URLs use more than one origin: ${[...origins].join(', ')}`);
|
|
}
|
|
|
|
/* ---------- the sitemap ---------- */
|
|
|
|
const sitemapPath = path.join(dist, 'sitemap.xml');
|
|
let sitemapCount = 0;
|
|
|
|
if (!existsSync(sitemapPath)) {
|
|
fail('(sitemap)', 'sitemap', 'sitemap.xml was not emitted');
|
|
} else {
|
|
const xml = await readFile(sitemapPath, 'utf8');
|
|
const blocks = [...xml.matchAll(/<url>([\s\S]*?)<\/url>/g)].map((m) => m[1]);
|
|
sitemapCount = blocks.length;
|
|
|
|
const listed = new Map();
|
|
for (const block of blocks) {
|
|
const loc = /<loc>([^<]*)<\/loc>/.exec(block)?.[1];
|
|
if (!loc) {
|
|
fail('(sitemap)', 'sitemap', 'a <url> block has no <loc>');
|
|
continue;
|
|
}
|
|
const route = new URL(loc).pathname.replace(/\/$/, '') || '/';
|
|
if (listed.has(route)) fail(route, 'sitemap', 'listed more than once');
|
|
|
|
const alternates = new Set(
|
|
[...block.matchAll(/<xhtml:link[^>]*hreflang="([^"]*)"/g)].map((m) => m[1]),
|
|
);
|
|
listed.set(route, alternates);
|
|
|
|
for (const hreflang of [...LOCALE_CODES, 'x-default']) {
|
|
if (!alternates.has(hreflang)) {
|
|
fail(route, 'sitemap', `entry is missing xhtml:link hreflang="${hreflang}"`);
|
|
}
|
|
}
|
|
}
|
|
|
|
/*
|
|
* The invariant, both ways round: every indexable page is listed, and nothing that
|
|
* asks not to be indexed is.
|
|
*
|
|
* Checking it here rather than trusting the two files to agree is the point. The
|
|
* page's robots meta and the sitemap are produced by different code from one shared
|
|
* predicate, and this reads the built output of both, so the day someone changes one
|
|
* without the other the build says so.
|
|
*/
|
|
for (const page of seen.values()) {
|
|
if (isNoindex(page)) {
|
|
if (listed.has(page.route)) {
|
|
fail(page.route, 'sitemap', 'says noindex but is listed in the sitemap');
|
|
}
|
|
continue;
|
|
}
|
|
if (!listed.has(page.route)) fail(page.route, 'sitemap', 'emitted but not in the sitemap');
|
|
}
|
|
for (const route of listed.keys()) {
|
|
if (!routes.has(route)) fail(route, 'sitemap', 'listed but never emitted');
|
|
}
|
|
}
|
|
|
|
/* ---------- report ---------- */
|
|
|
|
const localeCounts = LOCALE_CODES.map(
|
|
(code) => `${code}: ${[...seen.values()].filter((p) => p.locale === code).length}`,
|
|
).join(', ');
|
|
|
|
console.log(
|
|
`check-hreflang: ${pages.length} pages (${localeCounts}), ${sitemapCount} sitemap entries.`,
|
|
);
|
|
|
|
if (problems.length === 0) {
|
|
console.log('Every page declares its language, canonicalises to itself, and has a complete reciprocal alternate set.');
|
|
process.exit(0);
|
|
}
|
|
|
|
const grouped = new Map();
|
|
for (const problem of problems) {
|
|
if (!grouped.has(problem.kind)) grouped.set(problem.kind, []);
|
|
grouped.get(problem.kind).push(problem);
|
|
}
|
|
|
|
console.log(`\n${problems.length} problem(s):`);
|
|
for (const [kind, list] of grouped) {
|
|
console.log(`\n [${kind}] ${list.length}`);
|
|
for (const problem of list.slice(0, 15)) console.log(` ${problem.page}: ${problem.detail}`);
|
|
if (list.length > 15) console.log(` … and ${list.length - 15} more`);
|
|
}
|
|
process.exit(1);
|