Add Fedimint discovery, dual SQLite/Postgres storage, richer review handling, and generated social imagery.
136 lines
5.5 KiB
TypeScript
136 lines
5.5 KiB
TypeScript
/**
|
|
* The two indexing judgements the site makes, in one place.
|
|
*
|
|
* Both are imported by more than one file on purpose. `isIndexableMint` decides a mint
|
|
* page's `robots` meta *and* whether the sitemap lists it, and those two must agree:
|
|
* a sitemap entry for a page that says `noindex` is a contradiction a crawler notices.
|
|
* `check-hreflang.mjs` verifies the agreement from the built HTML, so the invariant is
|
|
* enforced rather than merely intended.
|
|
*
|
|
* Pure functions, no `t()`: `check-i18n.mjs` treats every `.ts` under `lib/` as island
|
|
* code and would hold any key here to the client-namespace rule.
|
|
*/
|
|
|
|
/**
|
|
* Is this mint's page worth a place in the index?
|
|
*
|
|
* No, in exactly one case: the site has never once reached it *and* nobody has ever
|
|
* reviewed it. Such a page has no operator name, no description, no version, no NUT
|
|
* list and no reviews, because every one of those comes from a mint that answered or a
|
|
* person who wrote something. What is left is the site's own furniture and a URL, which
|
|
* is the same page as the next never-reached mint's.
|
|
*
|
|
* It stays listed on /mints, stays linked, stays searchable on the site and stays
|
|
* reviewable — the rule that offline mints must remain findable is about this site's
|
|
* own listings, and nothing here changes them. Only search engines skip it, and only
|
|
* until the mint answers once or someone reviews it, at which point the next build
|
|
* puts it back.
|
|
*/
|
|
export function isIndexableMint(mint: {
|
|
last_online: number | null;
|
|
review_count: number;
|
|
}): boolean {
|
|
return !(mint.last_online === null && mint.review_count === 0);
|
|
}
|
|
|
|
/**
|
|
* Names that say nothing about which mint they belong to.
|
|
*
|
|
* Anchored, so a mint genuinely called "Bolverker Mint" or "Mint Cuba Bitcoin" is left
|
|
* alone; only a name that is *entirely* a generic noun matches. Two mints currently
|
|
* publish `name: "Cashu mint"`, which produced two identical `<title>`s competing with
|
|
* each other for the same query.
|
|
*/
|
|
const GENERIC_NAME = /^(cashu[\s-]?mint|mint|cashu|ecash[\s-]?mint|test(\s?mint)?)$/i;
|
|
|
|
/** Names published by more than one mint, lowercased. Built once from the full list. */
|
|
export function nameCollisions(mints: Array<{ name: string | null }>): Set<string> {
|
|
const seen = new Map<string, number>();
|
|
for (const mint of mints) {
|
|
const key = mint.name?.trim().toLowerCase();
|
|
if (!key) continue;
|
|
seen.set(key, (seen.get(key) ?? 0) + 1);
|
|
}
|
|
return new Set([...seen].filter(([, count]) => count > 1).map(([key]) => key));
|
|
}
|
|
|
|
/**
|
|
* The name a `<title>` should use for a mint.
|
|
*
|
|
* The operator's own name, unless it identifies nothing — either because it is missing,
|
|
* or because it is a generic noun, or because another mint publishes the same one. In
|
|
* those cases the domain does the identifying, since that is what actually tells two
|
|
* mints apart.
|
|
*
|
|
* This is for the title and the description only. The `<h1>` keeps the operator's name
|
|
* verbatim: the heading is the mint's name for itself, and rewriting it would be
|
|
* putting words in their mouth. The title is this site's label for its own page.
|
|
*/
|
|
export function seoName(
|
|
name: string | null,
|
|
domain: string,
|
|
collisions: Set<string>,
|
|
): string {
|
|
const raw = name?.trim() ?? '';
|
|
if (raw === '' || GENERIC_NAME.test(raw)) return domain;
|
|
if (collisions.has(raw.toLowerCase())) return `${raw} (${domain})`;
|
|
return raw;
|
|
}
|
|
|
|
/**
|
|
* Trim to a length without cutting a word in half.
|
|
*
|
|
* Meta descriptions are truncated by the search engine anyway, but a description stored
|
|
* ending mid-word reads as broken in every other consumer of it (link previews, feed
|
|
* readers) where nothing re-truncates.
|
|
*/
|
|
export function clampWords(value: string, max: number): string {
|
|
if (value.length <= max) return value;
|
|
const cut = value.slice(0, max);
|
|
const lastSpace = cut.lastIndexOf(' ');
|
|
return (lastSpace > max * 0.6 ? cut.slice(0, lastSpace) : cut).trimEnd();
|
|
}
|
|
|
|
/**
|
|
* A mint's own description, made fit to open a meta description.
|
|
*
|
|
* It is operator-written text headed for a `content` attribute and a link preview, so
|
|
* everything that does not belong in either comes out: HTML tags and control
|
|
* characters are stripped, whitespace collapses, and only the first sentence
|
|
* survives — a meta description quoting three sentences of marketing leaves no room
|
|
* for the rating and status that make the snippet useful. (Astro does the actual
|
|
* attribute escaping; this removes what escaping would merely neutralise.)
|
|
*
|
|
* Two judgement calls:
|
|
*
|
|
* - SHOUTING is normalised to sentence case. A description that is ≥80% capitals
|
|
* reads as noise in a search result, and the mint's page still shows the original.
|
|
* - The sentence split requires whitespace after the terminator, so "mint.example.com"
|
|
* inside a sentence does not end it.
|
|
*
|
|
* Returns null when nothing usable is left, and the caller falls back to its own copy.
|
|
*/
|
|
export function cleanMintDescription(value: string | null | undefined, max = 70): string | null {
|
|
if (!value) return null;
|
|
|
|
let clean = value
|
|
.replace(/<[^>]*>/g, ' ')
|
|
.replace(/[\u0000-\u001f\u007f-\u009f]/g, ' ')
|
|
.replace(/\s+/g, ' ')
|
|
.trim();
|
|
if (clean === '') return null;
|
|
|
|
const sentence = /^.*?[.!?](?=\s|$)/.exec(clean);
|
|
clean = (sentence?.[0] ?? clean).trim();
|
|
|
|
const letters = clean.replace(/[^\p{L}]/gu, '');
|
|
if (letters.length >= 8 && letters.replace(/\p{Ll}/gu, '').length / letters.length > 0.8) {
|
|
clean = clean.toLowerCase();
|
|
clean = clean[0]!.toUpperCase() + clean.slice(1);
|
|
}
|
|
|
|
clean = clampWords(clean, max);
|
|
if (!/[.!?]$/.test(clean)) clean += '.';
|
|
return clean;
|
|
}
|