/** * The two indexing judgements the site makes, in one place. * * Both are imported by more than one file on purpose. `isIndexableMint` decides a mint * page's `robots` meta *and* whether the sitemap lists it, and those two must agree: * a sitemap entry for a page that says `noindex` is a contradiction a crawler notices. * `check-hreflang.mjs` verifies the agreement from the built HTML, so the invariant is * enforced rather than merely intended. * * Pure functions, no `t()`: `check-i18n.mjs` treats every `.ts` under `lib/` as island * code and would hold any key here to the client-namespace rule. */ /** * Is this mint's page worth a place in the index? * * No, in exactly one case: the site has never once reached it *and* nobody has ever * reviewed it. Such a page has no operator name, no description, no version, no NUT * list and no reviews, because every one of those comes from a mint that answered or a * person who wrote something. What is left is the site's own furniture and a URL, which * is the same page as the next never-reached mint's. * * It stays listed on /mints, stays linked, stays searchable on the site and stays * reviewable — the rule that offline mints must remain findable is about this site's * own listings, and nothing here changes them. Only search engines skip it, and only * until the mint answers once or someone reviews it, at which point the next build * puts it back. */ export function isIndexableMint(mint: { last_online: number | null; review_count: number; }): boolean { return !(mint.last_online === null && mint.review_count === 0); } /** * Names that say nothing about which mint they belong to. * * Anchored, so a mint genuinely called "Bolverker Mint" or "Mint Cuba Bitcoin" is left * alone; only a name that is *entirely* a generic noun matches. Two mints currently * publish `name: "Cashu mint"`, which produced two identical ``s competing with * each other for the same query. */ const GENERIC_NAME = /^(cashu[\s-]?mint|mint|cashu|ecash[\s-]?mint|test(\s?mint)?)$/i; /** Names published by more than one mint, lowercased. Built once from the full list. */ export function nameCollisions(mints: Array<{ name: string | null }>): Set<string> { const seen = new Map<string, number>(); for (const mint of mints) { const key = mint.name?.trim().toLowerCase(); if (!key) continue; seen.set(key, (seen.get(key) ?? 0) + 1); } return new Set([...seen].filter(([, count]) => count > 1).map(([key]) => key)); } /** * The name a `<title>` should use for a mint. * * The operator's own name, unless it identifies nothing — either because it is missing, * or because it is a generic noun, or because another mint publishes the same one. In * those cases the domain does the identifying, since that is what actually tells two * mints apart. * * This is for the title and the description only. The `<h1>` keeps the operator's name * verbatim: the heading is the mint's name for itself, and rewriting it would be * putting words in their mouth. The title is this site's label for its own page. */ export function seoName( name: string | null, domain: string, collisions: Set<string>, ): string { const raw = name?.trim() ?? ''; if (raw === '' || GENERIC_NAME.test(raw)) return domain; if (collisions.has(raw.toLowerCase())) return `${raw} (${domain})`; return raw; } /** * Trim to a length without cutting a word in half. * * Meta descriptions are truncated by the search engine anyway, but a description stored * ending mid-word reads as broken in every other consumer of it (link previews, feed * readers) where nothing re-truncates. */ export function clampWords(value: string, max: number): string { if (value.length <= max) return value; const cut = value.slice(0, max); const lastSpace = cut.lastIndexOf(' '); return (lastSpace > max * 0.6 ? cut.slice(0, lastSpace) : cut).trimEnd(); } /** * A mint's own description, made fit to open a meta description. * * It is operator-written text headed for a `content` attribute and a link preview, so * everything that does not belong in either comes out: HTML tags and control * characters are stripped, whitespace collapses, and only the first sentence * survives — a meta description quoting three sentences of marketing leaves no room * for the rating and status that make the snippet useful. (Astro does the actual * attribute escaping; this removes what escaping would merely neutralise.) * * Two judgement calls: * * - SHOUTING is normalised to sentence case. A description that is ≥80% capitals * reads as noise in a search result, and the mint's page still shows the original. * - The sentence split requires whitespace after the terminator, so "mint.example.com" * inside a sentence does not end it. * * Returns null when nothing usable is left, and the caller falls back to its own copy. */ export function cleanMintDescription(value: string | null | undefined, max = 70): string | null { if (!value) return null; let clean = value .replace(/<[^>]*>/g, ' ') .replace(/[\u0000-\u001f\u007f-\u009f]/g, ' ') .replace(/\s+/g, ' ') .trim(); if (clean === '') return null; const sentence = /^.*?[.!?](?=\s|$)/.exec(clean); clean = (sentence?.[0] ?? clean).trim(); const letters = clean.replace(/[^\p{L}]/gu, ''); if (letters.length >= 8 && letters.replace(/\p{Ll}/gu, '').length / letters.length > 0.8) { clean = clean.toLowerCase(); clean = clean[0]!.toUpperCase() + clean.slice(1); } clean = clampWords(clean, max); if (!/[.!?]$/.test(clean)) clean += '.'; return clean; }