first commit

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
michilis
2026-08-20 22:41:25 +02:00
co-authored by Cursor
commit aa1771ea20
136 changed files with 27069 additions and 0 deletions
+300
View File
@@ -0,0 +1,300 @@
/**
* The i18n SEO check, over the built site.
*
* `check-links.mjs` answers "does every link go somewhere". This answers the question
* that matters once a site is in three languages: does every page correctly declare
* which language it is, which page it is, and where its siblings are. Those three
* declarations are what stop three translations of one page being indexed as three
* duplicates, and every one of them is easy to get subtly wrong in a way nothing else
* catches.
*
* Seven checks per page:
*
* lang <html lang> matches the locale its URL is under
* canonical present, absolute, and pointing at itself (never at another locale)
* alternates one per locale, plus exactly one x-default
* absolute every alternate href is absolute
* targets every alternate href is a page the build actually emitted
* reciprocal every alternate lists this page back, under this page's own hreflang
* x-default points at the default locale's version of this same page
*
* Plus, over the sitemap: every emitted page appears exactly once, with its own full
* xhtml:link alternate set.
*
* Usage: node scripts/check-hreflang.mjs [distDir]
* Exits non-zero on any problem, so it can gate a deploy.
*/
import { readdir, readFile } from 'node:fs/promises';
import { existsSync } from 'node:fs';
import path from 'node:path';
import { fileURLToPath } from 'node:url';
const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
const dist = path.resolve(process.argv[2] ?? path.join(root, 'dist'));
if (!existsSync(dist)) {
console.error(`No such directory: ${dist}. Run \`pnpm build\` first.`);
process.exit(2);
}
const { DEFAULT_LOCALE, LOCALE_CODES } = await import(path.join(root, 'src/i18n/locales.mjs'));
const PREFIXED = LOCALE_CODES.filter((code) => code !== DEFAULT_LOCALE);
/* ---------- what the build emitted ---------- */
async function walk(dir, base = dir) {
const out = [];
for (const entry of await readdir(dir, { withFileTypes: true })) {
const full = path.join(dir, entry.name);
if (entry.isDirectory()) out.push(...(await walk(full, base)));
else out.push(path.relative(base, full).split(path.sep).join('/'));
}
return out;
}
const files = await walk(dist);
const pages = files.filter((f) => f.endsWith('.html'));
/** dist-relative file to the route a visitor asks for. */
const routeOf = (file) =>
file === 'index.html' ? '/' : '/' + file.replace(/\/index\.html$/, '').replace(/\.html$/, '');
/** The locale a route is under, and the route with that prefix removed. */
function split(route) {
const first = route.split('/').filter(Boolean)[0];
if (first && PREFIXED.includes(first)) {
const rest = route.slice(first.length + 1);
return { locale: first, path: rest === '' ? '/' : rest };
}
return { locale: DEFAULT_LOCALE, path: route };
}
const problems = [];
const fail = (page, kind, detail) => problems.push({ page, kind, detail });
/* ---------- read every page ---------- */
const LANG = /<html[^>]*\slang="([^"]*)"/i;
const CANONICAL = /<link[^>]*rel="canonical"[^>]*href="([^"]*)"/i;
const ROBOTS = /<meta[^>]*name="robots"[^>]*content="([^"]*)"/i;
const ALTERNATE = /<link[^>]*rel="alternate"[^>]*hreflang="([^"]*)"[^>]*href="([^"]*)"/gi;
const seen = new Map(); // route -> { locale, path, lang, canonical, alternates: Map }
const routes = new Set();
for (const file of pages) {
const route = routeOf(file);
routes.add(route);
const html = await readFile(path.join(dist, file), 'utf8');
const alternates = new Map();
for (const match of html.matchAll(ALTERNATE)) alternates.set(match[1], match[2]);
seen.set(route, {
...split(route),
route,
lang: LANG.exec(html)?.[1] ?? null,
canonical: CANONICAL.exec(html)?.[1] ?? null,
robots: ROBOTS.exec(html)?.[1] ?? null,
alternates,
});
}
/*
* The 404 page is the one route with no place in the alternate graph.
*
* It is not a page anyone links to or a crawler should index; it is what a web server
* hands back for a URL that does not exist. It still gets a `lang` and it still gets
* one per locale, which is what the checks below verify, but it is deliberately absent
* from the sitemap and it is not required to be reciprocal with anything.
*/
const isOffGraph = (page) => page.path === '/404';
/**
* A page that exists but asks not to be indexed.
*
* Mint pages the site has never reached and nobody has reviewed carry this (see
* `src/lib/seo.ts`). Unlike the 404 they are real pages: they keep their canonical and
* their full alternate set, and only the sitemap leaves them out.
*/
const isNoindex = (page) => Boolean(page.robots?.includes('noindex'));
/* ---------- the seven checks ---------- */
const origins = new Set();
for (const page of seen.values()) {
if (page.lang !== page.locale) {
fail(page.route, 'lang', `<html lang="${page.lang}"> under /${page.locale}/`);
}
if (isOffGraph(page)) {
// No canonical, no alternates, and a robots meta saying so. Checked rather than
// skipped: "the 404 has no alternate set" is a claim worth verifying, because the
// easy mistake is for it to quietly acquire one.
if (page.canonical) fail(page.route, 'noindex', `404 has a canonical: ${page.canonical}`);
if (page.alternates.size > 0) {
fail(page.route, 'noindex', `404 declares ${page.alternates.size} alternate(s)`);
}
if (!page.robots?.includes('noindex')) fail(page.route, 'noindex', '404 is missing noindex');
continue;
}
if (!page.canonical) {
fail(page.route, 'canonical', 'no <link rel="canonical">');
continue;
}
if (!/^https?:\/\//.test(page.canonical)) {
fail(page.route, 'canonical', `relative: ${page.canonical}`);
continue;
}
const canonicalUrl = new URL(page.canonical);
origins.add(canonicalUrl.origin);
// Self-canonical. A locale canonicalising to another locale tells a crawler this
// page is a duplicate that should not be indexed, which is the exact opposite of
// what the alternate set beside it is claiming.
if (canonicalUrl.pathname.replace(/\/$/, '') !== page.route.replace(/\/$/, '')) {
fail(page.route, 'canonical', `points at ${canonicalUrl.pathname}, not at itself`);
}
// One per locale, plus x-default, and nothing else.
const expected = [...LOCALE_CODES, 'x-default'];
for (const hreflang of expected) {
if (!page.alternates.has(hreflang)) {
fail(page.route, 'alternates', `missing hreflang="${hreflang}"`);
}
}
for (const hreflang of page.alternates.keys()) {
if (!expected.includes(hreflang)) {
fail(page.route, 'alternates', `unexpected hreflang="${hreflang}"`);
}
}
for (const [hreflang, href] of page.alternates) {
if (!/^https?:\/\//.test(href)) {
fail(page.route, 'absolute', `hreflang="${hreflang}" is relative: ${href}`);
continue;
}
const target = new URL(href).pathname.replace(/\/$/, '') || '/';
if (!routes.has(target)) {
fail(page.route, 'targets', `hreflang="${hreflang}" points at ${target}, which was not emitted`);
continue;
}
if (hreflang === 'x-default') {
const expectedDefault = split(target);
if (expectedDefault.locale !== DEFAULT_LOCALE || expectedDefault.path !== page.path) {
fail(page.route, 'x-default', `points at ${target}, not at the ${DEFAULT_LOCALE} version of ${page.path}`);
}
continue;
}
// Reciprocity: the page this one points at has to point back, at this page, under
// this page's own hreflang. A set that is not reciprocal is discarded whole, so a
// one-sided link is worth less than no link.
const other = seen.get(target);
const back = other?.alternates.get(page.locale);
if (!back) {
fail(page.route, 'reciprocal', `${target} does not list hreflang="${page.locale}"`);
continue;
}
const backPath = new URL(back).pathname.replace(/\/$/, '') || '/';
if (backPath !== (page.route.replace(/\/$/, '') || '/')) {
fail(page.route, 'reciprocal', `${target} lists hreflang="${page.locale}" as ${backPath}`);
}
}
}
if (origins.size > 1) {
fail('(site)', 'origin', `canonical URLs use more than one origin: ${[...origins].join(', ')}`);
}
/* ---------- the sitemap ---------- */
const sitemapPath = path.join(dist, 'sitemap.xml');
let sitemapCount = 0;
if (!existsSync(sitemapPath)) {
fail('(sitemap)', 'sitemap', 'sitemap.xml was not emitted');
} else {
const xml = await readFile(sitemapPath, 'utf8');
const blocks = [...xml.matchAll(/<url>([\s\S]*?)<\/url>/g)].map((m) => m[1]);
sitemapCount = blocks.length;
const listed = new Map();
for (const block of blocks) {
const loc = /<loc>([^<]*)<\/loc>/.exec(block)?.[1];
if (!loc) {
fail('(sitemap)', 'sitemap', 'a <url> block has no <loc>');
continue;
}
const route = new URL(loc).pathname.replace(/\/$/, '') || '/';
if (listed.has(route)) fail(route, 'sitemap', 'listed more than once');
const alternates = new Set(
[...block.matchAll(/<xhtml:link[^>]*hreflang="([^"]*)"/g)].map((m) => m[1]),
);
listed.set(route, alternates);
for (const hreflang of [...LOCALE_CODES, 'x-default']) {
if (!alternates.has(hreflang)) {
fail(route, 'sitemap', `entry is missing xhtml:link hreflang="${hreflang}"`);
}
}
}
/*
* The invariant, both ways round: every indexable page is listed, and nothing that
* asks not to be indexed is.
*
* Checking it here rather than trusting the two files to agree is the point. The
* page's robots meta and the sitemap are produced by different code from one shared
* predicate, and this reads the built output of both, so the day someone changes one
* without the other the build says so.
*/
for (const page of seen.values()) {
if (isNoindex(page)) {
if (listed.has(page.route)) {
fail(page.route, 'sitemap', 'says noindex but is listed in the sitemap');
}
continue;
}
if (!listed.has(page.route)) fail(page.route, 'sitemap', 'emitted but not in the sitemap');
}
for (const route of listed.keys()) {
if (!routes.has(route)) fail(route, 'sitemap', 'listed but never emitted');
}
}
/* ---------- report ---------- */
const localeCounts = LOCALE_CODES.map(
(code) => `${code}: ${[...seen.values()].filter((p) => p.locale === code).length}`,
).join(', ');
console.log(
`check-hreflang: ${pages.length} pages (${localeCounts}), ${sitemapCount} sitemap entries.`,
);
if (problems.length === 0) {
console.log('Every page declares its language, canonicalises to itself, and has a complete reciprocal alternate set.');
process.exit(0);
}
const grouped = new Map();
for (const problem of problems) {
if (!grouped.has(problem.kind)) grouped.set(problem.kind, []);
grouped.get(problem.kind).push(problem);
}
console.log(`\n${problems.length} problem(s):`);
for (const [kind, list] of grouped) {
console.log(`\n [${kind}] ${list.length}`);
for (const problem of list.slice(0, 15)) console.log(` ${problem.page}: ${problem.detail}`);
if (list.length > 15) console.log(` … and ${list.length - 15} more`);
}
process.exit(1);