From 65307ba2780850e2c5729aa99b3174a5e272427e Mon Sep 17 00:00:00 2001 From: michilis Date: Tue, 25 Aug 2026 15:58:53 +0200 Subject: [PATCH 1/7] Compile the API instead of running its TypeScript in production. The unit's ExecStart named src/index.ts, so every start depended on the host having Node 22.18 or newer for native type stripping. A deploy onto a host with Node 20 met ERR_UNKNOWN_FILE_EXTENSION, exited in under a second, and was restarted 464 times over fifteen hours with nothing anywhere going red. api/tsconfig.json now emits to api/dist. The source keeps its explicit .ts import specifiers, which is what makes `node --watch src/index.ts` work in development; rewriteRelativeImportExtensions turns them into .js on the way out, so what runs in production is ordinary ESM that any Node from 20.18 up will start. `pnpm build` builds shared, then api, then web. `pnpm dev` is unchanged. deploy/ is tracked rather than ignored: the unit files are the thing an operator copies to /etc/systemd/system, and the alert unit added next has to live somewhere a deploy can find it. Co-Authored-By: Claude Opus 5 --- .gitignore | 1 - README.md | 55 +++++++++--- api/package.json | 3 +- api/tsconfig.json | 17 +++- deploy/cashumints-site.service | 61 +++++++++++++ deploy/cashumints-web.service | 90 +++++++++++++++++++ deploy/cashumints-web.timer | 25 ++++++ deploy/cashumints.service | 56 ++++++++++++ deploy/nginx.conf | 160 +++++++++++++++++++++++++++++++++ package.json | 4 +- 10 files changed, 455 insertions(+), 17 deletions(-) create mode 100644 deploy/cashumints-site.service create mode 100644 deploy/cashumints-web.service create mode 100644 deploy/cashumints-web.timer create mode 100644 deploy/cashumints.service create mode 100644 deploy/nginx.conf diff --git a/.gitignore b/.gitignore index fbdefa8..ac50c20 100644 --- a/.gitignore +++ b/.gitignore @@ -2,7 +2,6 @@ node_modules/ dist/ .astro/ api/data/ -deploy/ *.log .DS_Store .env diff --git a/README.md b/README.md index cb7d38e..8bd258b 100644 --- a/README.md +++ b/README.md @@ -52,9 +52,15 @@ rating encoding, and the bugs this rebuild fixes. ## Requirements -- Node 22.18 or newer (native TypeScript type stripping, so no build step for the API) +- Node 20.18 or newer to build and to run what a build produces +- Node 22.18 or newer to *develop*: `pnpm dev`, `pnpm seed` and the `api` test scripts + run `src/*.ts` through node directly, which needs native type stripping - pnpm 9 or newer +`engines.node` is the first of those, not the second, on purpose: it is the floor a +deployment has to clear, and a production host should never be told it needs a newer +Node than the compiled service actually runs on. + ## Setup ```bash @@ -286,7 +292,11 @@ pnpm typecheck pnpm build ``` -Output lands in `web/dist/`. Every page is prerendered once per language, so ~55 mints +Three packages in order, and the order is a dependency chain rather than a habit: +`shared` emits the types and the warning copy both other packages import, `api` compiles +`api/src` to `api/dist`, and `web` prerenders against a running API. + +Output lands in `api/dist/` and `web/dist/`. Every page is prerendered once per language, so ~55 mints and 9 static routes come out as ~200 pages, each with real titles, meta descriptions, OpenGraph and Twitter tags, a social card, a self-referencing canonical, a full hreflang set and a JSON-LD graph. `sitemap.xml` lists every indexable one with its `xhtml:link` @@ -782,6 +792,9 @@ literal specified behaviour. `shared/src/score.ts` carries the arithmetic, and ## Deployment +The unit files and the nginx block quoted below are checked in under `deploy/`. Those +are the copies to edit; what is quoted here is the same text, for reading in context. + Three units and an nginx block. Everything this project runs listens on loopback and runs as the same unprivileged user; nginx terminates TLS and proxies to it, and opens no file belonging to the project. @@ -806,19 +819,34 @@ is the one that built them. ### Node -The API and the site server both run TypeScript and ESM directly, with no build step, so -**systemd's node must be 22.18 or newer** — that is the release where native type -stripping stopped needing a flag. This is not the same question as `node -v` in your -shell: a version manager puts its node on the interactive `PATH` only, while systemd -resolves the absolute path in `ExecStart`. Check the one that matters: +**Node 20.18 or newer is enough.** Nothing systemd starts reads a `.ts` file: `pnpm +build` compiles `api/src` to `api/dist`, the site server is plain `.mjs`, and both units +run `/usr/bin/node` against ordinary JavaScript. 20.18 rather than 20.0 only because +both `ExecStart` lines pass `--env-file-if-exists`, which landed there. ```bash /usr/bin/node --version ``` -On Node 20 the API exits immediately with `ERR_UNKNOWN_FILE_EXTENSION` for `.ts` and -restarts forever. Install Node system-wide rather than pointing `ExecStart` at a version -manager's path, which breaks at the next upgrade and is invisible to `ProtectHome`. +That is the version that matters, and it is not the same question as `node -v` in your +shell: a version manager puts its node on the interactive `PATH` only, while systemd +resolves the absolute path in `ExecStart`. Install Node system-wide rather than pointing +`ExecStart` at a version manager's path, which breaks at the next upgrade and is +invisible to `ProtectHome`. + +**Why this section used to say 22.18.** The API ran `src/index.ts` directly, on native +type stripping, so the host's Node version was a runtime dependency of the service. A +deploy onto a host with Node 20 met `ERR_UNKNOWN_FILE_EXTENSION`, exited in under a +second, and was restarted by systemd 464 times over fifteen hours. Every dashboard was +green throughout, because there was no dashboard: `Restart=on-failure` with no start +limit is an infinite loop that never reports a failure. Two things changed. The service +is compiled, so the host's Node version cannot break it in that way again; and the units +now stop after five failures in two minutes and run an `OnFailure=` alert, so if +something else breaks it in some other way, the machine says so. See "Failing loudly". + +The version floor that is still 22.18 is the *development* one — `pnpm dev`, `pnpm seed`, +`pnpm migrate` and the `api` `test:*` scripts all hand `src/*.ts` to node. That is a +laptop requirement, not a server one. ### The API @@ -849,7 +877,10 @@ Environment=DB_PATH=/var/lib/cashumints/cashumints.db Environment=ICON_DIR=/var/lib/cashumints/icons # For Postgres, replace DB_PATH with DATABASE_URL and add After=postgresql.service. # Keep ICON_DIR either way: cached icons are files, not rows. -ExecStart=/usr/bin/node --env-file-if-exists=../.env src/index.ts + +# Compiled JavaScript, run by the distribution's own node. See "Node" above for why +# this is not src/index.ts any more. +ExecStart=/usr/bin/node --env-file-if-exists=../.env dist/index.js Restart=on-failure RestartSec=5s @@ -1161,7 +1192,7 @@ Order matters once: the site server refuses to start against a root that has no `index.html`, so the build has to publish before it comes up. ```bash -/usr/bin/node --version # 22.18 or newer, or the API will not run +/usr/bin/node --version # 20.18 or newer sudo systemctl enable --now cashumints # API first: the build reads from it sudo systemctl start cashumints-web # build, then publish to /var/lib/cashumints/web sudo systemctl enable --now cashumints-site # now it has something to serve diff --git a/api/package.json b/api/package.json index f61f2f4..ef471a4 100644 --- a/api/package.json +++ b/api/package.json @@ -4,8 +4,9 @@ "private": true, "type": "module", "scripts": { + "build": "tsc -p tsconfig.json", "dev": "node --env-file-if-exists=../.env --watch src/index.ts", - "start": "node --env-file-if-exists=../.env src/index.ts", + "start": "node --env-file-if-exists=../.env dist/index.js", "seed": "node --env-file-if-exists=../.env src/seed.ts", "migrate": "node --env-file-if-exists=../.env src/migrate.ts", "typecheck": "tsc -p tsconfig.json --noEmit", diff --git a/api/tsconfig.json b/api/tsconfig.json index 3389744..977249f 100644 --- a/api/tsconfig.json +++ b/api/tsconfig.json @@ -7,7 +7,22 @@ "strict": true, "noUncheckedIndexedAccess": true, "noImplicitOverride": true, - "noEmit": true, + /* + * This project is compiled now, rather than run straight off `src/*.ts`. + * + * The reason is a fifteen-hour crash loop: the unit's ExecStart named `src/index.ts`, + * the host's `/usr/bin/node` was 20, and native type stripping is 22.18 and newer, so + * every start died on ERR_UNKNOWN_FILE_EXTENSION and systemd restarted it 464 times + * without anything going red. Emitting plain `.js` removes the host's Node version + * from the set of things that can break a deploy. + * + * `rewriteRelativeImportExtensions` is what lets the source keep its explicit `.ts` + * specifiers — which is what makes `node --watch src/index.ts` work in development — + * while the emitted files import `./config.js` and run anywhere. + */ + "outDir": "dist", + "rootDir": "src", + "sourceMap": true, "allowImportingTsExtensions": true, "rewriteRelativeImportExtensions": true, "skipLibCheck": true, diff --git a/deploy/cashumints-site.service b/deploy/cashumints-site.service new file mode 100644 index 0000000..8a8a380 --- /dev/null +++ b/deploy/cashumints-site.service @@ -0,0 +1,61 @@ +# /etc/systemd/system/cashumints-site.service +# +# Serves the built site on loopback. nginx proxies to it and never opens a file itself, +# which is the point: when nginx held a `root` inside /home/cashumints, every directory +# down to dist had to be traversable by www-data, and the one that was not took the +# whole site down as a blanket 404 with nothing in the error log naming the cause. +# +# This is a long-running daemon, unlike cashumints-web.service next to it — that one is +# the oneshot that produces what this one serves. + +[Unit] +Description=cashumints.space static site server +Wants=network-online.target +After=network-online.target +# Give up after five failures in a minute instead of restarting forever. A process that +# cannot start will not start on the 4000th attempt either, and `failed` in +# `systemctl status` is a far louder signal than a journal scrolling past. These two are +# [Unit] keys; systemd ignores them under [Service] with only a warning. +StartLimitIntervalSec=60 +StartLimitBurst=5 +# Not Requires=cashumints.service: the pages are prerendered, so the site keeps serving +# a correct-as-of-last-build copy while the API is down. Only the islands go quiet. + +[Service] +Type=simple +User=cashumints +Group=cashumints +WorkingDirectory=/home/cashumints/CashuMints.space/web + +# The tree comes from cashumints-web.service, which rsyncs it here after a build. +# Serving web/dist directly would mean a rebuild empties the site for the length of it. +StateDirectory=cashumints +Environment=NODE_ENV=production +Environment=SITE_PORT=8789 +Environment=SITE_HOST=127.0.0.1 +Environment=WEB_ROOT=/var/lib/cashumints/web +ExecStart=/usr/bin/node server.mjs + +Restart=on-failure +RestartSec=5s +KillSignal=SIGTERM +# In-flight responses finish; idle keep-alive connections are closed at once. +TimeoutStopSec=15s +UMask=0027 + +NoNewPrivileges=true +PrivateTmp=true +PrivateDevices=true +ProtectSystem=strict +# Read-only rather than absent: server.mjs itself lives under /home/cashumints. +ProtectHome=read-only +ReadWritePaths=/var/lib/cashumints +ProtectKernelTunables=true +ProtectKernelModules=true +ProtectControlGroups=true +RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6 +RestrictSUIDSGID=true +LockPersonality=true + +[Install] +WantedBy=multi-user.target diff --git a/deploy/cashumints-web.service b/deploy/cashumints-web.service new file mode 100644 index 0000000..3718b0e --- /dev/null +++ b/deploy/cashumints-web.service @@ -0,0 +1,90 @@ +# /etc/systemd/system/cashumints-web.service +# +# The frontend is static: `output: 'static'` in astro.config.mjs, and the whole site is +# produced ahead of time. There is no frontend build to keep alive, so this unit is a +# build rather than a daemon — one shot of `pnpm build`, which compiles shared/, renders +# a social card per mint and prerenders every page from the live API. The daemon that +# hands the result out is cashumints-site.service. +# +# Run it after a deploy: +# sudo systemctl start cashumints-web +# +# Mint pages are prerendered, so new mints and new review counts only appear at the +# next build; pair this with a .timer for the nightly rebuild. There is deliberately no +# [Install] section — a rebuild should be scheduled, not fired on every boot. + +[Unit] +Description=Rebuild the cashumints.space static site +# Every page's data comes from the API over loopback, so the API has to be up. +# Requires= rather than Wants=: a dead API should abort the build, not replace a good +# site with an empty one. +Requires=cashumints.service +After=cashumints.service network-online.target +Wants=network-online.target + +[Service] +Type=oneshot +User=cashumints +Group=cashumints +WorkingDirectory=/home/cashumints/CashuMints.space + +# Where the published copy lands. Shared with the API and the site server, and created +# by systemd with this unit's ownership if it is not there yet. +StateDirectory=cashumints + +Environment=NODE_ENV=production +# Where the build reaches the API. Must match PORT= in cashumints.service. +Environment=API_URL=http://127.0.0.1:8788 +Environment=SITE_URL=https://cashumints.space +# Browser-facing origin. Empty means same origin: islands fetch /api/... and nginx +# forwards it. Set this only if the API ever moves to its own hostname. Declared here +# even though it is empty, because systemd's environment wins over .env — so what a +# production build emits cannot drift with an edit to that file. +Environment=PUBLIC_API_URL= + +# After= orders the start; it does not wait for the port to accept connections. At boot +# the API is still opening its database and probing, so block until it reports healthy +# rather than letting the first fetch die on ECONNREFUSED. /api/health answers 503 until +# it is genuinely ready, and curl -f treats that as a failure, so the loop keeps waiting. +ExecStartPre=/usr/bin/timeout 90 /bin/sh -c 'until curl -sf -o /dev/null http://127.0.0.1:8788/api/health; do sleep 1; done' +# Check `which pnpm` on the host: a corepack or pnpm-home install sits outside /usr/bin, +# and systemd's PATH does not include it. +ExecStart=/usr/bin/pnpm build + +# Publish, as a separate step from building. +# +# `astro build` empties dist before it writes, so the site server cannot read dist +# directly — a nightly rebuild would be a nightly minute of 404s. It serves this copy +# instead, and the copy is only touched once a build has succeeded: a failed build +# leaves the previous site up rather than replacing it with a half-written one, which is +# the same reason Requires=cashumints.service is above. +# +# --delay-updates stages the changed files and renames them in at the end, so the window +# where the tree is a mix of two builds is a rename rather than a whole transfer, and +# --delete-after keeps removals from landing before their replacements. Unchanged files +# — every hashed asset and card, which is nearly all of it — are not touched at all. +ExecStartPost=/usr/bin/rsync -a --delete-after --delay-updates web/dist/ /var/lib/cashumints/web/ + +# ~200 prerendered pages plus a card per mint. Minutes, not seconds, on a small VPS, and +# TimeoutStartSec is what bounds a Type=oneshot. +TimeoutStartSec=1800 +# A nightly rebuild should not starve the API it is reading from. +Nice=10 + +# The site server runs as cashumints and reads its own files, so this no longer has to +# be world-readable — it was 0022 for nginx, back when nginx opened the files as +# www-data. Kept at 0022 anyway: rsync preserves these modes into the published copy, +# and a readable static site is easier to inspect than one that needs sudo. +UMask=0022 + +NoNewPrivileges=true +PrivateTmp=true +PrivateDevices=true +# ProtectHome is deliberately absent, unlike in cashumints.service: this unit writes +# inside /home/cashumints — web/dist, web/public/og, web/src/generated and the pnpm +# store are all under it. +ProtectSystem=full +ProtectKernelTunables=true +ProtectKernelModules=true +ProtectControlGroups=true +RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6 diff --git a/deploy/cashumints-web.timer b/deploy/cashumints-web.timer new file mode 100644 index 0000000..c9a117e --- /dev/null +++ b/deploy/cashumints-web.timer @@ -0,0 +1,25 @@ +# /etc/systemd/system/cashumints-web.timer +# +# The nightly rebuild. Mint pages are prerendered, so a new mint or a new review count +# appears at the next build; between builds the site stays correct because the islands +# refresh status and reviews at runtime, and a mint indexed since the last build still +# resolves through the client-side fallback on the 404 page. +# +# sudo systemctl enable --now cashumints-web.timer +# +# The service it starts builds first and publishes second, so a failed build leaves the +# running site untouched rather than replacing it with a half-written one. + +[Unit] +Description=Nightly cashumints.space rebuild + +[Timer] +OnCalendar=*-*-* 03:30:00 +# Run it on the next boot if the machine was off at 03:30, rather than skipping a day. +Persistent=true +# Without this every host rebuilds on the same second. Harmless with one VPS; free +# insurance if there is ever a second. +RandomizedDelaySec=15m + +[Install] +WantedBy=timers.target diff --git a/deploy/cashumints.service b/deploy/cashumints.service new file mode 100644 index 0000000..3048edd --- /dev/null +++ b/deploy/cashumints.service @@ -0,0 +1,56 @@ +# /etc/systemd/system/cashumints.service +[Unit] +Description=cashumints.space indexer and API +Wants=network-online.target +After=network-online.target +# Stop after five failures in a minute rather than restarting forever. A wrong Node on +# PATH once produced four thousand identical crashes in the journal before anyone read +# one of them; `failed` in `systemctl status` says the same thing in one line. Both keys +# belong to [Unit] — under [Service] systemd only warns and ignores them. +StartLimitIntervalSec=60 +StartLimitBurst=5 + +[Service] +Type=simple +User=cashumints +Group=cashumints +WorkingDirectory=/home/cashumints/CashuMints.space/api + +# StateDirectory creates /var/lib/cashumints with the service user's ownership. +StateDirectory=cashumints +Environment=NODE_ENV=production +Environment=PORT=8788 +Environment=DB_PATH=/var/lib/cashumints/cashumints.db +Environment=ICON_DIR=/var/lib/cashumints/icons + +# Compiled JavaScript, run by the distribution's own node. +# +# This line used to read `src/index.ts`, which made every start depend on the host +# having Node 22.18 or newer for native type stripping. A host with Node 20 answered +# that with ERR_UNKNOWN_FILE_EXTENSION in under a second, 464 times over fifteen hours, +# and nothing anywhere went red. `pnpm build` now emits api/dist, so what runs here is +# ordinary ESM and any Node from 20.18 up will start it. +# +# Deliberately /usr/bin/node and nothing else: an nvm or fnm path is invisible to this +# unit's ProtectHome and breaks silently at the next version bump. +ExecStart=/usr/bin/node --env-file-if-exists=../.env dist/index.js + +Restart=on-failure +RestartSec=5s +KillSignal=SIGTERM +TimeoutStopSec=30s +UMask=0027 + +NoNewPrivileges=true +PrivateTmp=true +ProtectSystem=strict +ProtectHome=read-only +ReadWritePaths=/var/lib/cashumints +PrivateDevices=true +ProtectKernelTunables=true +ProtectKernelModules=true +ProtectControlGroups=true +RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6 + +[Install] +WantedBy=multi-user.target diff --git a/deploy/nginx.conf b/deploy/nginx.conf new file mode 100644 index 0000000..383c265 --- /dev/null +++ b/deploy/nginx.conf @@ -0,0 +1,160 @@ +# /etc/nginx/sites-available/cashumints.space +# +# nginx terminates TLS and proxies. It opens no file belonging to this project — not the +# built site, not an icon — and that is deliberate. +# +# It used to point a `root` at web/dist. Because nginx runs as www-data and everything +# this project owns runs as cashumints, that arrangement required every directory from / +# down to dist to be traversable by a user with no other business in the tree. A home +# directory at its default 0700 anywhere in that chain broke the entire site, and it +# broke it invisibly: `try_files` treats a permission error as a plain miss, so the +# symptom was a blanket 404, or an internal-redirect loop that ended in a 500 with the +# real cause named nowhere. +# +# Two upstreams now, both on loopback, both owned by the same user that built what they +# serve: +# +# 127.0.0.1:8789 cashumints-site.service the prerendered site +# 127.0.0.1:8788 cashumints.service /api/* and /icons/* +# +# Routing that used to live here lives with the thing that owns it. The locale 404 rule +# in particular was a hand-maintained alternation of 23 codes in a file that is not in +# the repository; adding a language meant remembering to edit it, and forgetting was +# silent. web/server.mjs resolves those from the built tree. + +proxy_cache_path /var/cache/nginx/cashumints + levels=1:2 + keys_zone=cashumints:10m + max_size=256m + inactive=10m + use_temp_path=off; + +# Keep a few connections open to each upstream rather than reconnecting per request. +# Both processes hold idle sockets longer than nginx does, so nginx is always the side +# that closes and there is no window where it reuses a socket the upstream just dropped +# — that race is what produces sporadic 502s under load. +upstream cashumints_site { + server 127.0.0.1:8789; + keepalive 16; +} + +upstream cashumints_api { + server 127.0.0.1:8788; + keepalive 8; +} + +server { + listen 80; + listen [::]:80; + server_name cashumints.space; + + location /.well-known/acme-challenge/ { + root /var/www/html; + } + + location / { + return 301 https://cashumints.space$request_uri; + } +} + +server { + listen 443 ssl http2; + listen [::]:443 ssl http2; + server_name cashumints.space; + + # nginx 1.25 and later want `http2 on;` on its own line and warn about the form above. + # Left as is because it is the form that works on both, and Debian 12 ships 1.22. + + ssl_certificate /etc/letsencrypt/live/cashumints.space/fullchain.pem; + ssl_certificate_key /etc/letsencrypt/live/cashumints.space/privkey.pem; + include /etc/letsencrypt/options-ssl-nginx.conf; + ssl_dhparam /etc/letsencrypt/ssl-dhparams.pem; + + # Set here rather than upstream: this is the only part of the stack that knows a + # request arrived over TLS. Add `preload` only once you are content never to serve + # this name over plain HTTP again. + add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always; + + # The upstreams send no Content-Encoding, so compression is nginx's to do. gzip_proxied + # any is required — without it nginx refuses to compress a proxied response at all. + gzip on; + gzip_proxied any; + gzip_vary on; + gzip_comp_level 5; + gzip_min_length 1024; + gzip_types text/plain text/css text/javascript application/javascript application/json + application/manifest+json application/xml image/svg+xml; + + # Nothing here accepts an upload. The API's largest body is an 8 KB JSON submission, + # and rejecting the oversized ones at the edge keeps them off the Node process. + client_max_body_size 16k; + + # Both upstreams are a process on this machine. A slow response is a bug, not a + # network condition, and failing fast beats holding a worker for a minute. + proxy_connect_timeout 2s; + proxy_read_timeout 30s; + proxy_send_timeout 30s; + + # HTTP/1.1 with an empty Connection header is what makes the keepalive pools above + # work; the default 1.0 opens a new socket per request. + proxy_http_version 1.1; + proxy_set_header Connection ""; + + # Every proxied location includes Debian's /etc/nginx/proxy_params, which sets Host, + # X-Real-IP, X-Forwarded-For and X-Forwarded-Proto. That include is load-bearing, not + # decorative: the API's rate limiter reads the last hop of X-Forwarded-For to tell two + # visitors apart, and without it every request arrives from the loopback peer and + # shares one bucket. On a distro that ships no proxy_params, set those four by hand. + # Do not also set X-Forwarded-For alongside the include — declaring it twice is what + # produces nginx's "could not build optimal proxy_headers_hash" warning. + + # The prerendered site. Cache-Control comes from server.mjs — a year and immutable for + # anything with a content hash in its name, revalidate-every-time for markup — so + # there is nothing to restate here. + location / { + proxy_pass http://cashumints_site; + include proxy_params; + } + + # Health must always reflect the live process. + location = /api/health { + proxy_pass http://cashumints_api; + proxy_cache off; + # add_header replaces rather than merges: declaring one here drops every add_header + # inherited from the server block, so HSTS has to be restated alongside it. + add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always; + add_header Cache-Control "no-store" always; + include proxy_params; + } + + # Briefly cache the read-heavy endpoints. + location ~ ^/api/(mints|stats)(?:/|$|\?) { + proxy_pass http://cashumints_api; + proxy_cache cashumints; + proxy_cache_valid 200 30s; + proxy_cache_valid 404 10s; + proxy_cache_use_stale updating error timeout http_500 http_502 http_503; + proxy_cache_background_update on; + proxy_cache_lock on; + # Restated for the same reason as in /api/health above. + add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always; + add_header X-Cache-Status $upstream_cache_status always; + include proxy_params; + } + + location /api/ { + proxy_pass http://cashumints_api; + include proxy_params; + } + + location /icons/ { + proxy_pass http://cashumints_api; + proxy_cache cashumints; + proxy_cache_valid 200 1d; + include proxy_params; + } + + # No error_page and no try_files. A miss is the site server's 404 page, in the right + # language and with a 404 status; an nginx error page here would replace it with a + # blank one and hide which upstream failed. +} diff --git a/package.json b/package.json index 890b9ac..edd1164 100644 --- a/package.json +++ b/package.json @@ -4,7 +4,7 @@ "version": "2.0.0", "type": "module", "engines": { - "node": ">=22.18" + "node": ">=20.18" }, "scripts": { "dev": "pnpm --parallel --filter ./api --filter ./web dev", @@ -12,7 +12,7 @@ "dev:web": "pnpm --filter ./web dev", "seed": "pnpm --filter ./api seed", "bones": "pnpm --filter ./web bones", - "build": "pnpm --filter ./shared build && pnpm --filter ./web build", + "build": "pnpm --filter ./shared build && pnpm --filter ./api build && pnpm --filter ./web build", "start": "pnpm --filter ./web start", "check:links": "pnpm --filter ./web check:links", "typecheck": "pnpm -r typecheck", From 9ffa53094d9a00c216e3c3a521db0c6f3414a946 Mon Sep 17 00:00:00 2001 From: michilis Date: Tue, 25 Aug 2026 16:07:01 +0200 Subject: [PATCH 2/7] Make a starved discovery cycle say so, in the log and on /api/health. MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit For about a year the production RELAYS list did not include the relay carrying the kind 38000/38172 archive. Every backfill read about thirty events, wrote them faithfully, reported ok=true, and the nightly build republished an index of eight mints. Nothing measured the difference between "the cycle completed" and "the cycle read anything", so nothing went red. Three signals now do: - Per-relay attribution. queryRelays() replaces pool.querySync(), which merges every relay into one deduplicated array and throws away who sent what. It keeps one subscription per relay over the pool's existing sockets and shares a single alreadyHaveEvent across them, so an event five relays carry is still verified once; receivedEvent fires before that check, which is what makes the per-relay count mean "what this relay contributed". The deadline moved out of each Subscription's own EOSE timer so `eose` means a frame arrived rather than something timed out. - A WARN naming any relay that will not connect, on every cycle, and any relay that connected and sent nothing, on backfills only. An incremental cycle is supposed to come back empty. - BACKFILL_MIN_EVENTS, default 200. Under it, ERROR discovery starvation suspected and a flag health reports as discovery_starved, forcing 503. Sticky across incremental cycles so an hourly cycle finding four events cannot clear what a backfill diagnosed; stored in the database so a restart cannot either. A fresh database is starved until its first backfill lands. That is intended: it holds the build's health gate rather than publishing a site made from nothing. Verified against the live relay set — 1528 events, five relays connected, EOSE on all five, health 200 — and against an unreachable list, which produces the two WARN lines, the ERROR, and 503. Co-Authored-By: Claude Opus 5 --- .env.example | 13 ++ README.md | 79 +++++++++++- api/src/config.ts | 14 +++ api/src/discovery.ts | 290 ++++++++++++++++++++++++++++++++++++++++--- api/src/queries.ts | 34 ++++- shared/src/types.ts | 43 +++++++ 6 files changed, 455 insertions(+), 18 deletions(-) diff --git a/.env.example b/.env.example index f4d6fa1..2b735a0 100644 --- a/.env.example +++ b/.env.example @@ -96,6 +96,19 @@ SEO_PRODUCT_JSONLD=1 # so reviews the old site published to snort/primal were invisible to it. RELAYS=wss://relay.cashumints.space,wss://nos.lol,wss://relay.azzamo.net,wss://relay.snort.social,wss://relay.primal.net +# How many events a backfill has to read before it counts as having read anything. +# +# A backfill asks every relay above for the whole history of four kinds; on a working +# relay list that is thousands of events. Under this floor, discovery logs +# `ERROR discovery starvation suspected` and /api/health answers 503 with +# `discovery_starved: true` until the next backfill clears it. +# +# This exists because a RELAYS list missing the relay that carries the announcement +# archive returned about thirty events per backfill for a year, reported ok=true every +# time, and left the index at eight mints with every health signal green. Lower it only +# for a private or test relay that genuinely holds less; 1 disables the check. +#BACKFILL_MIN_EVENTS=200 + # Profile relays for the BUILD (kind 0, prerendered reviewer names on the home # page). A wider pool than RELAYS on purpose: relay.cashumints.space holds no kind # 0 at all and snort/primal hold almost none, so the two aggregators below are what diff --git a/README.md b/README.md index 8bd258b..0b66e34 100644 --- a/README.md +++ b/README.md @@ -585,6 +585,7 @@ so a systemd `Environment=` line or a one-off `PORT=9000 pnpm dev:api` still ove | `DB_POOL_MAX` | `10` | Postgres connections held open. Unused by SQLite. | | `ICON_DIR` | `api/data/icons` | Cached mint icons, served at `/icons/*` | | `RELAYS` | see `shared/src/nostr.ts` | Comma separated relay list | +| `BACKFILL_MIN_EVENTS` | `200` | Events a backfill has to read before it counts as one. Under it, discovery logs `ERROR discovery starvation suspected` and health goes 503. See [Starvation](#discovery-starvation). | | `PROBE_INTERVAL_MIN` | `10` | Minutes between probe cycles | | `DISCOVERY_INTERVAL_MIN` | `60` | Minutes between discovery cycles | | `PROBE_CONCURRENCY` | `8` | Mints probed in parallel | @@ -717,7 +718,7 @@ Five endpoints, CORS open, no auth. Four read; the fifth writes. | Endpoint | Notes | | -------------------- | ------------------------------------------------------------------ | -| `GET /api/health` | Never cached. 503 when probes are stale or discovery failed. | +| `GET /api/health` | Never cached. 503 when probes are stale, discovery failed, or discovery is starved. Carries the last cycle's per-relay outcome. | | `GET /api/stats` | Network counters, memoized 60s in process. | | `GET /api/mints` | Everything listed, online first then score descending. `?limit=`, `?type=`. | | `GET /api/mints/:host` | One listing plus its ecosystem's own fields, distribution, uptime and probe history. | @@ -725,6 +726,82 @@ Five endpoints, CORS open, no auth. Four read; the fifth writes. `/icons/*` serves the cached mint icons. +### Discovery starvation + +For about a year, `GET /api/health` said `ok`, every discovery cycle reported +`ok=true`, and the index sat at eight mints. The production `RELAYS` list did not +include the relay carrying the kind 38000/38172 archive, so each backfill read about +thirty events, wrote them faithfully, and the nightly build republished the result. +Nothing was broken in a way anything measured. + +What was missing is that "the cycle completed" and "the cycle read anything" are +different claims, and only the first one was being made. Three things now make the +second one: + +**Per-relay attribution.** A cycle records, for every relay in `RELAYS`, whether a +socket opened, how many events it sent, and whether it ended in a real EOSE. Counts are +taken before cross-relay deduplication, so they say what each relay contributed rather +than what happened to be new because of it. The one-line cycle log carries the lot: + +``` +INFO discovery cycle mode=backfill events=1528 … starved=false \ + relays=wss://relay.cashumints.space=12 wss://nos.lol=1566 wss://relay.azzamo.net=10 \ + wss://relay.snort.social=18 wss://relay.primal.net=1 +``` + +Read that line before changing `RELAYS`. It is also how you find out that most of this +network's archive currently sits behind one relay. + +**A WARN per relay, naming it.** A relay that would not connect is warned about on every +cycle. A relay that connected and sent nothing is warned about on backfills only — an +incremental cycle asking for one interval is *supposed* to come back empty, and an +hourly warning about that would train everyone to skip the line that eventually matters. + +``` +WARN discovery relay unreachable relay=wss://relay.example.invalid mode=backfill +WARN discovery relay returned no events relay=wss://relay.azzamo.net mode=backfill +``` + +**A floor.** `BACKFILL_MIN_EVENTS`, 200 by default. A backfill asks five relays for the +entire history of four kinds; on a working relay set that is thousands of events. Under +the floor: + +``` +ERROR discovery starvation suspected events=31 floor=200 relays=5 silent=4 unreachable=0 +``` + +and a flag is set that `GET /api/health` reports as `discovery_starved`, which forces +`status` to `degraded` and the response to **503**. The flag is sticky across +incremental cycles: an hourly cycle finding four events must not clear a starvation a +backfill diagnosed. Only the next backfill clears it. + +`GET /api/health` grew four fields for this: + +```json +{ + "status": "degraded", + "discovery_relays": [ + { "url": "wss://relay.cashumints.space", "connected": true, "events": 12, "eose": true }, + { "url": "wss://relay.example.invalid", "connected": false, "events": 0, "eose": false } + ], + "last_discovery_events": 31, + "last_discovery_mode": "backfill", + "discovery_starved": true, + "backfill_min_events": 200 +} +``` + +**A fresh database reports 503 until its first backfill finishes, and that is correct.** +Before any backfill has run, nothing has confirmed that this deployment's relay list +reads anything at all, and answering `ok` would be the original bug in miniature. In +practice it holds `cashumints-web.service` at its health gate — which is the point: a +first deploy should not publish a site built from an empty index. The state is stored in +the database rather than in memory for the same reason, so a restart cannot launder a +starvation into "no cycle yet". + +To silence it deliberately on a deployment that genuinely has less history than this — +a private relay, a test rig — set `BACKFILL_MIN_EVENTS=1`. + ### Indexing on demand ```bash diff --git a/api/src/config.ts b/api/src/config.ts index f242b17..e573ce3 100644 --- a/api/src/config.ts +++ b/api/src/config.ts @@ -112,6 +112,20 @@ export const config = { iconDir: process.env['ICON_DIR'] ?? path.join(apiRoot, 'data', 'icons'), relays: (process.env['RELAYS']?.split(',').map((r) => r.trim()).filter(Boolean) ?? [...DEFAULT_RELAYS]) as string[], + /** + * The floor a backfill cycle has to clear before it counts as a real read. + * + * For about a year this deployment's RELAYS list did not include the relay carrying + * the kind 38000/38172 archive. Every backfill returned about thirty events, wrote + * them, reported ok=true, and the index sat at eight mints while every health signal + * stayed green. A backfill asks five relays for the whole history of four kinds; on a + * working relay set it comes back with thousands. Anything under this is not a quiet + * network, it is a misconfigured one, and it says so in the log and on /api/health. + * + * Raise it on a deployment that genuinely has more history, lower it for a local + * test relay. It is deliberately not zero-able: set it to 1 if you mean "off". + */ + backfillMinEvents: int('BACKFILL_MIN_EVENTS', 200), probeIntervalMin: int('PROBE_INTERVAL_MIN', 10), discoveryIntervalMin: int('DISCOVERY_INTERVAL_MIN', 60), probeConcurrency: int('PROBE_CONCURRENCY', 8), diff --git a/api/src/discovery.ts b/api/src/discovery.ts index d39000a..e3a1531 100644 --- a/api/src/discovery.ts +++ b/api/src/discovery.ts @@ -19,9 +19,10 @@ import { type FedimintAnnouncement, type LnurlAnnouncement, type LnurlFields, + type RelayHealth, } from '@cashumints/shared'; import { config } from './config.ts'; -import { getDb, setState, getStateNumber } from './db.ts'; +import { getDb, setState, getState, getStateNumber } from './db.ts'; import type { Sql } from './db-driver.ts'; import { log } from './log.ts'; import { insertMintIfNew, upsertFedimint, upsertLnurl } from './mints.ts'; @@ -39,8 +40,118 @@ export interface DiscoveryResult { newMints: string[]; newReviews: number; ok: boolean; + /** What each configured relay actually did, in `config.relays` order. */ + relays: RelayHealth[]; + /** A backfill that came in under `config.backfillMinEvents`. */ + starved: boolean; } +/** + * What the last cycle did, kept so /api/health can answer for it. + * + * Written to the `state` table rather than held in memory, because the question it + * answers — "is discovery actually reading anything?" — has to survive the restart that + * would otherwise reset it to "no cycle yet, nothing to report". A process that crash + * loops would clear an in-memory flag on every attempt. + */ +export interface DiscoveryReport { + at: number; + mode: 'backfill' | 'incremental'; + events: number; + ok: boolean; + relays: RelayHealth[]; + starved: boolean; +} + +/** `state` key holding the JSON of the above. */ +const REPORT_KEY = 'last_discovery_report'; + +/** + * The last cycle's report, or null before any cycle has run. + * + * A row that will not parse reads as null — the same as no cycle — because the caller + * is a health endpoint and "I cannot tell you" must not be dressed up as "fine". + */ +export async function lastDiscoveryReport(): Promise { + const raw = await getState(REPORT_KEY); + if (!raw) return null; + try { + const parsed = JSON.parse(raw) as DiscoveryReport; + return Array.isArray(parsed.relays) ? parsed : null; + } catch { + return null; + } +} + +/** + * Per-relay bookkeeping for one cycle. + * + * Every relay in `config.relays` gets a row up front, including the ones that are never + * reached, because a relay that produced no row at all is exactly the one worth naming: + * the year-long starvation was a relay list that connected cleanly and simply did not + * hold the archive, and the only field that would have shown it is a zero here. + * + * `events` counts what a relay sent *before* cross-relay deduplication, so five relays + * carrying the same 400 events report 400 each rather than 400 once and 0 four times. + * Attribution is the whole point; the deduplicated total is reported separately. + */ +class RelayTally { + private readonly rows = new Map(); + + constructor(urls: readonly string[]) { + for (const url of urls) { + this.rows.set(url, { events: 0, subs: 0, eoses: 0, connected: false }); + } + } + + private row(url: string) { + let found = this.rows.get(url); + if (!found) { + found = { events: 0, subs: 0, eoses: 0, connected: false }; + this.rows.set(url, found); + } + return found; + } + + connected(url: string): void { + this.row(url).connected = true; + } + + subscribed(url: string): void { + this.row(url).subs++; + } + + event(url: string): void { + this.row(url).events++; + } + + eose(url: string): void { + this.row(url).eoses++; + } + + /** One row per configured relay, in configuration order. */ + list(): RelayHealth[] { + return [...this.rows.entries()].map(([url, row]) => ({ + url, + connected: row.connected, + events: row.events, + // A cycle asks a relay many questions. It only counts as having reached the end + // of the stream if it reached the end of every one of them. + eose: row.subs > 0 && row.eoses === row.subs, + })); + } +} + +/** + * Long enough that the relay's own EOSE timer never wins. + * + * `Subscription` fires `oneose` both when an EOSE frame arrives and when its internal + * timer expires, so the two are indistinguishable from the callback. Pushing that timer + * out of reach and running the deadline here instead is what makes `eose` in the report + * mean "the relay said it was done" rather than "something gave up". + */ +const NEVER_EOSE_MS = 24 * 60 * 60 * 1000; + let pool: SimplePool | null = null; /** @@ -66,12 +177,100 @@ export function closePool(): void { pool = null; } +/** + * Ask every configured relay one filter, and record what each of them did. + * + * This replaces `pool.querySync(config.relays, …)`, which answers the same question and + * throws the attribution away: it merges five relays into one deduplicated array, so a + * relay list where four relays are empty and one carries everything is indistinguishable + * from five healthy ones. That indistinguishability is the bug this whole file is being + * changed for — a year of ~31-event backfills, `ok=true` every time. + * + * What it keeps from `querySync`, deliberately: + * + * - One subscription per relay over the pool's existing sockets, so this is the same + * number of connections as before. + * - A single `alreadyHaveEvent` shared across all five. `AbstractRelay._onmessage` + * consults it *before* `JSON.parse` and signature verification, so an event five + * relays all carry is still verified once. Per-relay `querySync` calls would have + * verified it five times, which at 500 events a page is real CPU. + * - `receivedEvent`, which fires on the way past that check, so the per-relay count is + * what the relay sent rather than what was new because of it. + * + * What it changes: the deadline is run here rather than by each `Subscription`'s own + * EOSE timer, so `oneose` firing means an EOSE frame actually arrived. See NEVER_EOSE_MS. + * + * Never throws. A relay that will not connect is a fact to record, not a reason to + * abandon the four that did. + */ +async function queryRelays(filter: Filter, tally: RelayTally | null): Promise { + const events: NostrEvent[] = []; + const known = new Set(); + const alreadyHaveEvent = (id: string): boolean => { + if (known.has(id)) return true; + known.add(id); + return false; + }; + + await Promise.all( + config.relays.map(async (url) => { + let relay; + try { + // The same connection budget subscribeMap would have used for this maxWait. + relay = await getPool().ensureRelay(url, { + connectionTimeout: Math.max(MAX_WAIT_MS * 0.8, MAX_WAIT_MS - 1000), + }); + } catch { + // Left as connected=false in the tally, which is the whole report this needs. + return; + } + tally?.connected(url); + + await new Promise((resolve) => { + let settled = false; + let deadline: ReturnType | undefined; + const finish = (): void => { + if (settled) return; + settled = true; + if (deadline !== undefined) clearTimeout(deadline); + resolve(); + }; + + try { + const sub = relay.subscribe([filter], { + onevent: (event) => events.push(event), + alreadyHaveEvent, + receivedEvent: () => tally?.event(url), + oneose: () => { + tally?.eose(url); + sub.close('closed automatically on eose'); + }, + onclose: finish, + eoseTimeout: NEVER_EOSE_MS, + }); + tally?.subscribed(url); + deadline = setTimeout(() => sub.close('closed on maxWait'), MAX_WAIT_MS); + } catch { + // The socket went away between ensureRelay and the REQ. + finish(); + } + }); + }), + ); + + return events; +} + /** * Query one kind, paging backwards with `until` until a page yields nothing new. * Relays cap `limit` independently, so paging is the only way a fresh database * converges to the complete history. */ -async function fetchKind(kind: number, since: number | null): Promise { +async function fetchKind( + kind: number, + since: number | null, + tally: RelayTally | null, +): Promise { const seen = new Map(); let until: number | undefined; @@ -82,7 +281,7 @@ async function fetchKind(kind: number, since: number | null): Promise { const filters: Filter[] = []; const base: Filter = { kinds: [KIND_REVIEW], limit: QUERY_LIMIT }; @@ -178,11 +378,7 @@ async function fetchReviewsForMint( } const batches = await Promise.all( - filters.map((filter) => - getPool() - .querySync(config.relays, filter, { maxWait: MAX_WAIT_MS }) - .catch(() => [] as NostrEvent[]), - ), + filters.map((filter) => queryRelays(filter, tally).catch(() => [] as NostrEvent[])), ); return batches.flat(); @@ -596,6 +792,7 @@ export async function runDiscovery(backfill: boolean): Promise const since = lastRun === null ? null : Math.max(0, lastRun - 3600); const newMints = new Set(); + const tally = new RelayTally(config.relays); let newReviews = 0; let events = 0; let ok = true; @@ -610,7 +807,7 @@ export async function runDiscovery(backfill: boolean): Promise const announcementsByType = new Map(); await Promise.all( Object.entries(ANNOUNCEMENT_KINDS).map(async ([type, kind]) => { - announcementsByType.set(type, await fetchKind(kind, since)); + announcementsByType.set(type, await fetchKind(kind, since, tally)); }), ); @@ -625,10 +822,10 @@ export async function runDiscovery(backfill: boolean): Promise // Announcements alone miss mints that only ever appear in a review's `u` tag, // so reviews feed discovery too. - const reviews = await fetchKind(KIND_REVIEW, since); + const reviews = await fetchKind(KIND_REVIEW, since, tally); // The recent window catches anything a relay dropped from the unbounded query. const recent = - since === null ? await fetchKind(KIND_REVIEW, now - RECENT_WINDOW_S) : []; + since === null ? await fetchKind(KIND_REVIEW, now - RECENT_WINDOW_S, tally) : []; const byId = new Map(); for (const e of [...reviews, ...recent]) byId.set(e.id, e); @@ -689,7 +886,7 @@ export async function runDiscovery(backfill: boolean): Promise while (cursor < targets.length) { const target = targets[cursor++]; if (!target) continue; - const found = await fetchReviewsForMint(target, since); + const found = await fetchReviewsForMint(target, since, tally); if (found.length > 0) { events += found.length; newReviews += await ingestReviews(found, index); @@ -707,14 +904,79 @@ export async function runDiscovery(backfill: boolean): Promise log.error('discovery failed', { reason: err instanceof Error ? err.message : String(err) }); } + const mode = backfill ? 'backfill' : 'incremental'; + const relays = tally.list(); + + /* + * Name the relay, every time, one line each. + * + * A relay that would not connect is worth saying on any cycle: the address is wrong, + * or it is down, and neither gets better by itself. A relay that connected and sent + * nothing is only news on a backfill — an incremental cycle asking for the last hour + * of four kinds legitimately comes back empty, and warning about that hourly would + * train everyone to skip the line that eventually matters. + */ + for (const relay of relays) { + if (!relay.connected) { + log.warn('discovery relay unreachable', { relay: relay.url, mode }); + continue; + } + if (backfill && relay.events === 0) { + log.warn('discovery relay returned no events', { relay: relay.url, mode }); + } else if (!relay.eose) { + log.warn('discovery relay never reached EOSE', { + relay: relay.url, + mode, + events: relay.events, + }); + } + } + + /* + * The floor, and the flag the health endpoint reads. + * + * Only a backfill is measured against it. A backfill asks for the entire history of + * every announcement kind and every review, so on a working relay set it is thousands + * of events; an incremental cycle asks for one interval and is supposed to be small. + * + * The flag is sticky across incremental cycles: an hourly cycle that finds four + * events must not clear a starvation a backfill diagnosed, so a non-backfill carries + * forward whatever the last backfill concluded. + */ + let starved: boolean; + if (backfill) { + starved = events < config.backfillMinEvents; + if (starved) { + log.error('discovery starvation suspected', { + events, + floor: config.backfillMinEvents, + relays: relays.length, + silent: relays.filter((r) => r.events === 0).length, + unreachable: relays.filter((r) => !r.connected).length, + hint: 'check RELAYS: a relay list missing the announcement archive looks exactly like this', + }); + } + } else { + // No backfill has ever run in this deployment: nothing has confirmed the relay set + // reads anything, and saying "fine" would be the whole original bug. + starved = (await lastDiscoveryReport())?.starved ?? true; + } + + const report: DiscoveryReport = { at: now, mode, events, ok, relays, starved }; + // A report that cannot be written is not worth failing a cycle over; the cycle's own + // work is already committed, and health degrades on the stale timestamp instead. + await setState(REPORT_KEY, JSON.stringify(report)).catch(() => undefined); + log.info('discovery cycle', { - mode: backfill ? 'backfill' : 'incremental', + mode, events, new_mints: newMints.size, new_reviews: newReviews, ok, + starved, + relays: relays.map((r) => `${r.url}=${r.connected ? r.events : 'down'}`).join(' '), ms: Date.now() - started, }); - return { events, newMints: [...newMints], newReviews, ok }; + return { events, newMints: [...newMints], newReviews, ok, relays, starved }; } diff --git a/api/src/queries.ts b/api/src/queries.ts index 59df339..0d59b46 100644 --- a/api/src/queries.ts +++ b/api/src/queries.ts @@ -16,6 +16,7 @@ import { } from '@cashumints/shared'; import { config, startedAt } from './config.ts'; import { getDb, getStateNumber, getState } from './db.ts'; +import { lastDiscoveryReport } from './discovery.ts'; import { mintByHost, parseEcosystem, type MintRow } from './mints.ts'; /** @@ -379,28 +380,55 @@ export function resetStatsCache(): void { statsCache = null; } -/** Health bypasses the stats cache: it is the endpoint you page on. */ +/** + * Health bypasses the stats cache: it is the endpoint you page on. + * + * Three things can degrade it, and they are three different failures: + * + * probeStale nothing has checked a mint in three intervals + * !discoveryOk the last discovery cycle threw + * report.starved the last backfill read less than BACKFILL_MIN_EVENTS + * + * The third is the one added after the postmortem, and it is the only one that would + * have caught a year of the index sitting at eight mints: the cycles were completing, + * `ok` was true, the probes were fresh, and the relay list simply did not contain the + * relay holding the archive. "Ran without throwing" is not the same claim as "read + * anything", and only the second one is worth a green light. + * + * A deployment with an empty database reports degraded until its first backfill lands, + * because until then nothing has confirmed the relay set reads anything at all. That is + * intended: it holds `cashumints-web.service` at its health gate rather than letting it + * publish a site built from nothing. + */ export async function getHealth(): Promise { const now = Math.floor(Date.now() / 1000); const db = await getDb(); - const [lastProbe, lastDiscovery, discoveryOkRaw, tracked] = await Promise.all([ + const [lastProbe, lastDiscovery, discoveryOkRaw, tracked, report] = await Promise.all([ getStateNumber('last_probe_at'), getStateNumber('last_discovery_at'), getState('last_discovery_ok'), db.get<{ n: number }>('SELECT COUNT(*) AS n FROM mints'), + lastDiscoveryReport(), ]); const discoveryOk = discoveryOkRaw !== '0'; const staleAfter = config.probeIntervalMin * 60 * 3; const probeStale = lastProbe === null || now - lastProbe > staleAfter; + // No report at all is starvation by default: see the note above. + const starved = report?.starved ?? true; return { - status: probeStale || !discoveryOk ? 'degraded' : 'ok', + status: probeStale || !discoveryOk || starved ? 'degraded' : 'ok', uptime_s: now - startedAt, last_probe_at: lastProbe, last_discovery_at: lastDiscovery, mints_tracked: tracked?.n ?? 0, updated_at: now, + discovery_relays: report?.relays ?? [], + last_discovery_events: report?.events ?? null, + last_discovery_mode: report?.mode ?? null, + discovery_starved: starved, + backfill_min_events: config.backfillMinEvents, }; } diff --git a/shared/src/types.ts b/shared/src/types.ts index 4177c24..d8691f1 100644 --- a/shared/src/types.ts +++ b/shared/src/types.ts @@ -236,6 +236,29 @@ export interface Stats { lnurl_reviews: number; } +/** + * How one configured relay answered during the last discovery cycle. + * + * The three facts are deliberately separate, because the failure this exists to catch + * had all three looking different from each other: the relays in `RELAYS` connected + * fine, reached EOSE fine, and simply did not carry the archive, so `events` was the + * only field that would have said anything. A relay that is down and a relay that is + * up and empty are different problems with different fixes. + */ +export interface RelayHealth { + url: string; + /** A socket was opened to it. false means the address is wrong or the relay is down. */ + connected: boolean; + /** + * Events it sent, counted before cross-relay deduplication — so this is what *this* + * relay contributed, not what was new because of it. Zero on a connected relay is + * the interesting number. + */ + events: number; + /** Every query it was asked ended in a real EOSE rather than in a timeout. */ + eose: boolean; +} + /** `GET /api/health`. */ export interface Health { status: 'ok' | 'degraded'; @@ -244,6 +267,26 @@ export interface Health { last_discovery_at: number | null; mints_tracked: number; updated_at: number; + /** + * Per-relay outcome of the last discovery cycle. Empty until one has run — including + * on a fresh database, which is why a brand new deployment reports degraded until its + * first backfill finishes. + */ + discovery_relays: RelayHealth[]; + /** Unique events the last discovery cycle received. null before the first one. */ + last_discovery_events: number | null; + /** Which kind of cycle those numbers describe. */ + last_discovery_mode: 'backfill' | 'incremental' | null; + /** + * The last backfill came back under `backfill_min_events`, or none has run yet. + * + * This is the flag that would have caught a year of ~31-event backfills against a + * relay list missing the archive. It forces `status` to degraded, and /api/health to + * 503, which is what the build gate and the site's own health checks read. + */ + discovery_starved: boolean; + /** `BACKFILL_MIN_EVENTS`, echoed so a reader of this payload can see the threshold. */ + backfill_min_events: number; } /** From 0ebc8ada54dc0124acef4c723dd68ade898898ea Mon Sep 17 00:00:00 2001 From: michilis Date: Tue, 25 Aug 2026 16:10:01 +0200 Subject: [PATCH 3/7] Make a crash loop reach somebody instead of scrolling past. MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `Restart=on-failure` with no start limit is an infinite loop by definition: the unit never reaches `failed`, `systemctl status` stays active (auto-restart), and the only evidence is a journal moving at four lines a second. That is how 464 restarts over fifteen hours went unnoticed. All three units now stop after five failures in 120s and run OnFailure=cashumints-alert@%n.service. The window is 120s and not 60s because RestartSec=5s plus a process that takes a few seconds to die can spread five failures past a sixty second window, reset the counter, and loop forever anyway. cashumints-alert@.service is a oneshot that takes the failed unit's name as its instance. Configuration is /etc/cashumints/alert.env: NTFY_URL gets a plain-text body, WEBHOOK_URL gets JSON carrying `content` so one payload fits Discord and Slack-compatible endpoints. With neither set — or the file absent — it still writes to the journal at ERROR via a `<3>` syslog prefix, so `journalctl -p err -t cashumints-alert` is a complete history on a host nobody configured. It cannot become a second thing to debug: each curl is bounded at 10s, each failure falls back to a journal line, and the shell ends in `true`, so the alerter always exits 0. Verified with systemd-analyze verify and by running the ExecStart body against a local sink — the JSON parses, and every branch exits 0. Co-Authored-By: Claude Opus 5 --- README.md | 193 ++++++++++++++++++++++++++++++- deploy/alert.env.example | 23 ++++ deploy/cashumints-alert@.service | 101 ++++++++++++++++ deploy/cashumints-site.service | 16 ++- deploy/cashumints-web.service | 5 + deploy/cashumints.service | 24 +++- 6 files changed, 347 insertions(+), 15 deletions(-) create mode 100644 deploy/alert.env.example create mode 100644 deploy/cashumints-alert@.service diff --git a/README.md b/README.md index 0b66e34..f29dbc3 100644 --- a/README.md +++ b/README.md @@ -933,12 +933,15 @@ laptop requirement, not a server one. Description=cashumints.space indexer and API Wants=network-online.target After=network-online.target -# Stop after five failures in a minute rather than restarting forever: a process that -# cannot start will not start on the 4000th attempt either, and `failed` in +# Stop after five failures in two minutes rather than restarting forever: a process +# that cannot start will not start on the 4000th attempt either, and `failed` in # `systemctl status` is a louder signal than a journal scrolling past. Both keys belong -# to [Unit] — under [Service] systemd only warns and ignores them. -StartLimitIntervalSec=60 +# to [Unit] — under [Service] systemd only warns and ignores them. See "Failing loudly" +# for why the window is 120s and not 60s. +StartLimitIntervalSec=120 StartLimitBurst=5 +# And carry that `failed` off the machine. %n is this unit's own name. +OnFailure=cashumints-alert@%n.service [Service] Type=simple @@ -1009,8 +1012,9 @@ Two behaviours are worth knowing about because they are load-bearing: Description=cashumints.space static site server Wants=network-online.target After=network-online.target -StartLimitIntervalSec=60 +StartLimitIntervalSec=120 StartLimitBurst=5 +OnFailure=cashumints-alert@%n.service # Not Requires=cashumints.service: the pages are prerendered, so the site keeps serving # a correct-as-of-last-build copy while the API is down. Only the islands go quiet. @@ -1175,6 +1179,184 @@ nginx 1.25 and later want `http2 on;` on its own line and warn about the `listen form above; Debian 12 ships 1.22, where the newer form is an unknown directive. The form above is the one that works on both. +### Failing loudly + +Two separate silences produced today's incident, and they need separate fixes. + +The first is a **crash loop that never reports a failure**. `Restart=on-failure` with no +start limit is an infinite loop by definition: systemd restarts, the process dies, +systemd restarts. The unit never reaches `failed`, so `systemctl status` stays `active +(auto-restart)`, nothing sends anything anywhere, and the only evidence is a journal +scrolling past at four lines a second. The API did this 464 times over fifteen hours. + +All three units now carry: + +```ini +StartLimitIntervalSec=120 +StartLimitBurst=5 +OnFailure=cashumints-alert@%n.service +``` + +**Why 120 and not 60.** `RestartSec=5s` means five attempts cost a little over twenty +seconds of waiting, plus however long each attempt survives before dying. A process that +fails *slowly* — a database connection that times out, a port that takes four seconds to +refuse — spreads five failures past a sixty second window, resets the counter, and loops +forever anyway. 120s covers the slow case. Both keys belong under `[Unit]`; systemd +takes them under `[Service]` with only a warning and then ignores them. + +**Why `OnFailure=` at all.** `StartLimitBurst` turns the loop into a `failed` state, +which is much better, and is still a state somebody has to go and look at. `OnFailure=` +is what makes the machine speak first. `%n` expands to the failed unit's own name, which +arrives as the template instance in `%i`. + +The second silence is a **build that publishes an index it should have refused**; that +one is the `MIN_MINTS_FOR_BUILD` gate under "Rebuilds", and the `discovery_starved` flag +under "Discovery starvation" is what feeds it. + +#### The alert unit + +```ini +# /etc/systemd/system/cashumints-alert@.service +# +# The unit that makes a failure audible. +# +# The other three units each carry `OnFailure=cashumints-alert@%n.service`, so systemd +# starts one of these with the failed unit's name as the instance — `%i` below is +# literally `cashumints.service`, `cashumints-web.service` or `cashumints-site.service`. +# +# Why it exists: the API once crash looped 464 times over fifteen hours and nothing said +# so. `Restart=on-failure` with no start limit is an infinite loop that never reaches a +# `failed` state, so the journal filled with identical lines nobody was reading and +# every signal stayed green. The other half of the fix is StartLimitBurst= in each unit, +# which turns the loop into a failure; this is what carries that failure off the machine. +# +# Install: +# sudo install -m 0644 deploy/cashumints-alert@.service /etc/systemd/system/ +# sudo install -d -m 0755 /etc/cashumints +# sudo install -m 0640 -o root -g root deploy/alert.env.example /etc/cashumints/alert.env +# sudo systemctl daemon-reload +# +# No [Install] section and never enabled: OnFailure= starts it, and a unit that also +# started at boot would page on every reboot. + +[Unit] +Description=Notify that %i failed +# No OnFailure= here. An alerter that alerts about its own failure is a loop, and this +# one is written so its worst case is a journal line rather than a retry. + +[Service] +Type=oneshot + +# The one file an operator edits, and the only reason this unit is configurable at all. +# Absent is a supported state — the leading `-` says so — and then the ExecStart below +# still writes to the journal at ERROR, which is what `systemctl status` and +# `journalctl -p err` read. See alert.env.example. +EnvironmentFile=-/etc/cashumints/alert.env + +# So `journalctl -t cashumints-alert` finds every alert, whichever unit triggered it. +SyslogIdentifier=cashumints-alert + +# Everything is inside one shell so the "nothing configured" branch is reachable without +# a second unit. The pieces, in order: +# +# - `printf '<3>…'` on stdout. systemd reads that syslog prefix off a journal stream +# and files the line at priority 3, ERROR, so `journalctl -p err` is a complete +# history of failures on a host with no webhook configured at all. `<4>` is warning. +# A prefix rather than systemd-cat, so the unit needs nothing from the filesystem it +# has just sandboxed itself away from. +# - NTFY_URL is a topic URL (https://ntfy.sh/your-topic). It gets a plain-text body +# naming the failed unit, plus the header names ntfy understands. +# - WEBHOOK_URL gets a JSON POST instead, for Discord, Slack or anything that speaks +# `{"content": …}` — every key is sent, so one payload fits all of them. +# - `--max-time 10` and a `||` fallback on each: an alert that hangs would hold the +# failed unit's job open, and an alert that fails must not itself become a second +# failed unit for somebody to notice. The shell ends in `true` for the same reason. +# +# `%i` is the failed unit's name, passed as an argument rather than interpolated into +# the shell text: systemd expands specifiers before /bin/sh ever sees the line, and a +# unit name is not a thing to trust to quoting. +ExecStart=/bin/sh -c '\ + UNIT="$1"; \ + HOST="$(hostname)"; \ + WHEN="$(date -Is)"; \ + TEXT="$UNIT failed on $HOST at $WHEN"; \ + printf "<3>%s\\n" "$TEXT"; \ + if [ -n "$NTFY_URL" ]; then \ + /usr/bin/curl -fsS --max-time 10 \ + -H "Title: cashumints: $UNIT failed" \ + -H "Priority: high" \ + -H "Tags: rotating_light" \ + -d "$TEXT" "$NTFY_URL" >/dev/null \ + || printf "<3>%s\\n" "alert: POST to NTFY_URL failed"; \ + fi; \ + if [ -n "$WEBHOOK_URL" ]; then \ + /usr/bin/curl -fsS --max-time 10 \ + -H "Content-Type: application/json" \ + -d "{\\"unit\\":\\"$UNIT\\",\\"host\\":\\"$HOST\\",\\"at\\":\\"$WHEN\\",\\"text\\":\\"$TEXT\\",\\"content\\":\\"$TEXT\\"}" \ + "$WEBHOOK_URL" >/dev/null \ + || printf "<3>%s\\n" "alert: POST to WEBHOOK_URL failed"; \ + fi; \ + if [ -z "$NTFY_URL" ] && [ -z "$WEBHOOK_URL" ]; then \ + printf "<4>%s\\n" "alert: no NTFY_URL or WEBHOOK_URL in /etc/cashumints/alert.env, journal only"; \ + fi; \ + true' _ %i + +# It sends one HTTP request and writes one line. It needs no identity of its own, and +# DynamicUser gives it a throwaway one rather than sharing `nobody` with everything else +# on the host that also could not be bothered to make a user. +DynamicUser=yes +NoNewPrivileges=true +PrivateDevices=true +ProtectSystem=strict +ProtectHome=true +ProtectKernelTunables=true +ProtectKernelModules=true +ProtectControlGroups=true +RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6 +RestrictSUIDSGID=true +LockPersonality=true +# An alert that cannot reach the network in ten seconds is not worth a stuck job. +TimeoutStartSec=30 +``` + +Configuration is one optional file. With it absent, or with both values empty, a failure +still lands in the journal at `ERROR` and is readable with `journalctl -p err -t +cashumints-alert`; the unit is written so that its worst case is a log line rather than +a second thing to debug. + +```ini +# /etc/cashumints/alert.env +# +# Read by cashumints-alert@.service, which systemd starts when any of the three units +# fails. Everything here is optional: with the file absent or both values empty, an +# alert is still written to the journal at ERROR priority and is readable with +# +# journalctl -p err -t cashumints-alert +# +# Set one or both to have failures leave the machine. +# +# Install it root-owned and not world-readable — a webhook URL is a capability: +# sudo install -d -m 0755 /etc/cashumints +# sudo install -m 0640 -o root -g root deploy/alert.env.example /etc/cashumints/alert.env +# sudo systemctl daemon-reload + +# An ntfy topic URL. Free and public at ntfy.sh; pick a topic name nobody will guess, +# because anyone who knows it can read and post to it. +#NTFY_URL=https://ntfy.sh/cashumints-alerts-CHANGE-ME + +# Anything that accepts a JSON POST. The body carries `unit`, `host`, `at`, `text` and +# `content` — the last of which is what Discord and most Slack-compatible endpoints read, +# so one payload fits all three. +#WEBHOOK_URL=https://discord.com/api/webhooks/… +``` + +Test it without breaking anything: + +```bash +sudo systemctl start 'cashumints-alert@test.service' +journalctl -t cashumints-alert -n 5 --no-pager +``` + ### Rebuilds Mint pages are prerendered, so new mints and new review counts appear at the next build. @@ -1196,6 +1378,7 @@ Description=Rebuild the cashumints.space static site Requires=cashumints.service After=cashumints.service network-online.target Wants=network-online.target +OnFailure=cashumints-alert@%n.service [Service] Type=oneshot diff --git a/deploy/alert.env.example b/deploy/alert.env.example new file mode 100644 index 0000000..3f6b638 --- /dev/null +++ b/deploy/alert.env.example @@ -0,0 +1,23 @@ +# /etc/cashumints/alert.env +# +# Read by cashumints-alert@.service, which systemd starts when any of the three units +# fails. Everything here is optional: with the file absent or both values empty, an +# alert is still written to the journal at ERROR priority and is readable with +# +# journalctl -p err -t cashumints-alert +# +# Set one or both to have failures leave the machine. +# +# Install it root-owned and not world-readable — a webhook URL is a capability: +# sudo install -d -m 0755 /etc/cashumints +# sudo install -m 0640 -o root -g root deploy/alert.env.example /etc/cashumints/alert.env +# sudo systemctl daemon-reload + +# An ntfy topic URL. Free and public at ntfy.sh; pick a topic name nobody will guess, +# because anyone who knows it can read and post to it. +#NTFY_URL=https://ntfy.sh/cashumints-alerts-CHANGE-ME + +# Anything that accepts a JSON POST. The body carries `unit`, `host`, `at`, `text` and +# `content` — the last of which is what Discord and most Slack-compatible endpoints read, +# so one payload fits all three. +#WEBHOOK_URL=https://discord.com/api/webhooks/… diff --git a/deploy/cashumints-alert@.service b/deploy/cashumints-alert@.service new file mode 100644 index 0000000..fb9a1bd --- /dev/null +++ b/deploy/cashumints-alert@.service @@ -0,0 +1,101 @@ +# /etc/systemd/system/cashumints-alert@.service +# +# The unit that makes a failure audible. +# +# The other three units each carry `OnFailure=cashumints-alert@%n.service`, so systemd +# starts one of these with the failed unit's name as the instance — `%i` below is +# literally `cashumints.service`, `cashumints-web.service` or `cashumints-site.service`. +# +# Why it exists: the API once crash looped 464 times over fifteen hours and nothing said +# so. `Restart=on-failure` with no start limit is an infinite loop that never reaches a +# `failed` state, so the journal filled with identical lines nobody was reading and +# every signal stayed green. The other half of the fix is StartLimitBurst= in each unit, +# which turns the loop into a failure; this is what carries that failure off the machine. +# +# Install: +# sudo install -m 0644 deploy/cashumints-alert@.service /etc/systemd/system/ +# sudo install -d -m 0755 /etc/cashumints +# sudo install -m 0640 -o root -g root deploy/alert.env.example /etc/cashumints/alert.env +# sudo systemctl daemon-reload +# +# No [Install] section and never enabled: OnFailure= starts it, and a unit that also +# started at boot would page on every reboot. + +[Unit] +Description=Notify that %i failed +# No OnFailure= here. An alerter that alerts about its own failure is a loop, and this +# one is written so its worst case is a journal line rather than a retry. + +[Service] +Type=oneshot + +# The one file an operator edits, and the only reason this unit is configurable at all. +# Absent is a supported state — the leading `-` says so — and then the ExecStart below +# still writes to the journal at ERROR, which is what `systemctl status` and +# `journalctl -p err` read. See alert.env.example. +EnvironmentFile=-/etc/cashumints/alert.env + +# So `journalctl -t cashumints-alert` finds every alert, whichever unit triggered it. +SyslogIdentifier=cashumints-alert + +# Everything is inside one shell so the "nothing configured" branch is reachable without +# a second unit. The pieces, in order: +# +# - `printf '<3>…'` on stdout. systemd reads that syslog prefix off a journal stream +# and files the line at priority 3, ERROR, so `journalctl -p err` is a complete +# history of failures on a host with no webhook configured at all. `<4>` is warning. +# A prefix rather than systemd-cat, so the unit needs nothing from the filesystem it +# has just sandboxed itself away from. +# - NTFY_URL is a topic URL (https://ntfy.sh/your-topic). It gets a plain-text body +# naming the failed unit, plus the header names ntfy understands. +# - WEBHOOK_URL gets a JSON POST instead, for Discord, Slack or anything that speaks +# `{"content": …}` — every key is sent, so one payload fits all of them. +# - `--max-time 10` and a `||` fallback on each: an alert that hangs would hold the +# failed unit's job open, and an alert that fails must not itself become a second +# failed unit for somebody to notice. The shell ends in `true` for the same reason. +# +# `%i` is the failed unit's name, passed as an argument rather than interpolated into +# the shell text: systemd expands specifiers before /bin/sh ever sees the line, and a +# unit name is not a thing to trust to quoting. +ExecStart=/bin/sh -c '\ + UNIT="$1"; \ + HOST="$(hostname)"; \ + WHEN="$(date -Is)"; \ + TEXT="$UNIT failed on $HOST at $WHEN"; \ + printf "<3>%s\\n" "$TEXT"; \ + if [ -n "$NTFY_URL" ]; then \ + /usr/bin/curl -fsS --max-time 10 \ + -H "Title: cashumints: $UNIT failed" \ + -H "Priority: high" \ + -H "Tags: rotating_light" \ + -d "$TEXT" "$NTFY_URL" >/dev/null \ + || printf "<3>%s\\n" "alert: POST to NTFY_URL failed"; \ + fi; \ + if [ -n "$WEBHOOK_URL" ]; then \ + /usr/bin/curl -fsS --max-time 10 \ + -H "Content-Type: application/json" \ + -d "{\\"unit\\":\\"$UNIT\\",\\"host\\":\\"$HOST\\",\\"at\\":\\"$WHEN\\",\\"text\\":\\"$TEXT\\",\\"content\\":\\"$TEXT\\"}" \ + "$WEBHOOK_URL" >/dev/null \ + || printf "<3>%s\\n" "alert: POST to WEBHOOK_URL failed"; \ + fi; \ + if [ -z "$NTFY_URL" ] && [ -z "$WEBHOOK_URL" ]; then \ + printf "<4>%s\\n" "alert: no NTFY_URL or WEBHOOK_URL in /etc/cashumints/alert.env, journal only"; \ + fi; \ + true' _ %i + +# It sends one HTTP request and writes one line. It needs no identity of its own, and +# DynamicUser gives it a throwaway one rather than sharing `nobody` with everything else +# on the host that also could not be bothered to make a user. +DynamicUser=yes +NoNewPrivileges=true +PrivateDevices=true +ProtectSystem=strict +ProtectHome=true +ProtectKernelTunables=true +ProtectKernelModules=true +ProtectControlGroups=true +RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6 +RestrictSUIDSGID=true +LockPersonality=true +# An alert that cannot reach the network in ten seconds is not worth a stuck job. +TimeoutStartSec=30 diff --git a/deploy/cashumints-site.service b/deploy/cashumints-site.service index 8a8a380..131e321 100644 --- a/deploy/cashumints-site.service +++ b/deploy/cashumints-site.service @@ -12,12 +12,18 @@ Description=cashumints.space static site server Wants=network-online.target After=network-online.target -# Give up after five failures in a minute instead of restarting forever. A process that -# cannot start will not start on the 4000th attempt either, and `failed` in -# `systemctl status` is a far louder signal than a journal scrolling past. These two are -# [Unit] keys; systemd ignores them under [Service] with only a warning. -StartLimitIntervalSec=60 +# Give up after five failures in two minutes instead of restarting forever. A process +# that cannot start will not start on the 4000th attempt either, and `failed` in +# `systemctl status` is a far louder signal than a journal scrolling past. The window +# matches cashumints.service; see the note there for why it is 120s and not 60s. These +# two are [Unit] keys; systemd ignores them under [Service] with only a warning. +StartLimitIntervalSec=120 StartLimitBurst=5 + +# Carry a failure off the machine. `%n` is this unit's own name, so the alert says +# which one died. cashumints-alert@.service writes to the journal at ERROR always and +# curls NTFY_URL or WEBHOOK_URL from /etc/cashumints/alert.env when either is set. +OnFailure=cashumints-alert@%n.service # Not Requires=cashumints.service: the pages are prerendered, so the site keeps serving # a correct-as-of-last-build copy while the API is down. Only the islands go quiet. diff --git a/deploy/cashumints-web.service b/deploy/cashumints-web.service index 3718b0e..efe9020 100644 --- a/deploy/cashumints-web.service +++ b/deploy/cashumints-web.service @@ -22,6 +22,11 @@ Requires=cashumints.service After=cashumints.service network-online.target Wants=network-online.target +# Carry a failure off the machine. `%n` is this unit's own name, so the alert says +# which one died. cashumints-alert@.service writes to the journal at ERROR always and +# curls NTFY_URL or WEBHOOK_URL from /etc/cashumints/alert.env when either is set. +OnFailure=cashumints-alert@%n.service + [Service] Type=oneshot User=cashumints diff --git a/deploy/cashumints.service b/deploy/cashumints.service index 3048edd..495ebb9 100644 --- a/deploy/cashumints.service +++ b/deploy/cashumints.service @@ -3,13 +3,27 @@ Description=cashumints.space indexer and API Wants=network-online.target After=network-online.target -# Stop after five failures in a minute rather than restarting forever. A wrong Node on -# PATH once produced four thousand identical crashes in the journal before anyone read -# one of them; `failed` in `systemctl status` says the same thing in one line. Both keys -# belong to [Unit] — under [Service] systemd only warns and ignores them. -StartLimitIntervalSec=60 +# Stop after five failures in two minutes rather than restarting forever. +# +# The window is 120s and not 60s because RestartSec=5s below means five attempts take +# a little over twenty seconds of restarts plus however long each attempt lives before +# it dies. A process that fails *slowly* — a database that times out, a port that takes +# four seconds to refuse — can spread five failures past a sixty second window and reset +# the counter forever, which is the loop this is supposed to stop. 120s covers that. +# +# The failure this exists for: ExecStart named a .ts file, /usr/bin/node was 20, and +# every start died in under a second. 464 restarts over fifteen hours, and because +# Restart=on-failure without a start limit never reaches a `failed` state, nothing +# anywhere went red. Both keys belong to [Unit] — under [Service] systemd only warns and +# ignores them. +StartLimitIntervalSec=120 StartLimitBurst=5 +# Carry a failure off the machine. `%n` is this unit's own name, so the alert says +# which one died. cashumints-alert@.service writes to the journal at ERROR always and +# curls NTFY_URL or WEBHOOK_URL from /etc/cashumints/alert.env when either is set. +OnFailure=cashumints-alert@%n.service + [Service] Type=simple User=cashumints From 060c7f1a59c98121e5ef7872c40602fcd7b5538e Mon Sep 17 00:00:00 2001 From: michilis Date: Tue, 25 Aug 2026 16:13:16 +0200 Subject: [PATCH 4/7] Refuse to publish a site built from a hollow index. MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Health answering 200 and the index being complete are different claims. A year of ~31-event backfills left a perfectly healthy API serving a real, correct, complete list of eight mints. A build against that succeeds — it prerenders eight cards — and rsync --delete-after then replaces fifty-five with eight. cashumints-web.service gains a second ExecStartPre after the health wait: count /api/mints, and exit non-zero below MIN_MINTS_FOR_BUILD (default 20, overridable with `systemctl edit`). A refusal aborts the unit before `pnpm build`, and publishing is ExecStartPost, so the previously published site is untouched; the OnFailure alert added in the last commit says why. Counted by the "host": key rather than by counting braces, because the list payload is about to carry a nested object per mint. A curl that fails at all counts as zero, which is below every floor — so an API that fell over between the health check and this line refuses the build instead of sailing through it. Verified against three live APIs: 73 mints passes, a doctored 8-mint database fails with the reason, and a dead port fails. Co-Authored-By: Claude Opus 5 --- README.md | 51 +++++++++++++++++++++++++++++++++++ deploy/cashumints-web.service | 34 +++++++++++++++++++++++ 2 files changed, 85 insertions(+) diff --git a/README.md b/README.md index f29dbc3..e2bc9f4 100644 --- a/README.md +++ b/README.md @@ -1368,6 +1368,23 @@ Publishing is a separate step from building, and the separation is the point: th the site server reads is only touched once a build has succeeded, so a failed build leaves the previous site up rather than replacing it with a half-written one. +**Two gates before the build starts.** The first waits for `/api/health` to answer 200, +because `After=` orders a start and does not wait for a port. The second counts +`/api/mints` and refuses to build below `MIN_MINTS_FOR_BUILD`, default 20. + +The second exists because health answering 200 and the index being complete are +different claims. A year of ~31-event backfills left a perfectly healthy API serving a +real, correct, complete list of eight mints; a build against that succeeds, prerenders +eight cards, and `rsync --delete-after` replaces fifty-five with eight. The gate runs as +an `ExecStartPre`, so a refusal aborts the unit before `pnpm build` — and publishing is +`ExecStartPost`, after the build — which means the previously published site is never +touched. The `OnFailure=` alert says why. + +```bash +# Raise or lower it for this host without editing the unit: +sudo systemctl edit cashumints-web # [Service] / Environment=MIN_MINTS_FOR_BUILD=40 +``` + ```ini # /etc/systemd/system/cashumints-web.service [Unit] @@ -1401,6 +1418,40 @@ Environment=PUBLIC_API_URL= # rather than letting the first fetch die on ECONNREFUSED. /api/health answers 503 until # it is genuinely ready, and curl -f treats that as a failure, so the loop keeps waiting. ExecStartPre=/usr/bin/timeout 90 /bin/sh -c 'until curl -sf -o /dev/null http://127.0.0.1:8788/api/health; do sleep 1; done' + +# Then: does the API actually have an index to build a site out of? +# +# Health answering 200 says the process is up and its last backfill read something. It +# does not say how many mints are in the table, and those are different questions — the +# year of ~31-event backfills had a healthy API serving a real, complete, correct list of +# eight mints. A build against that succeeds, prerenders eight cards, and rsync happily +# replaces fifty-five with eight. +# +# So count the list before spending twenty minutes building from it. Below the floor +# this exits non-zero, systemd abandons the unit at ExecStartPre, and — because publishing +# is ExecStartPost, after the build — the previously published site is never touched. The +# site stays exactly as it was and the OnFailure alert says why. +# +# Counted by the `"host":` key, one per item, rather than by counting `{`: the list +# payload carries a nested object per mint (its NUT capability switches), so brace +# counting would report roughly double. No jq: it is not installed on this host and a +# build gate should not add a dependency to run. +# +# `Q` is a double-quote character, built with printf rather than written literally, +# because this whole command is already inside systemd's single quotes and a quote of +# either kind in the grep pattern would end the argument early. +# +# A curl that fails for any reason leaves `n` empty, `$${n:-0}` reads that as zero, and +# zero is below every floor — so an API that fell over between the health check above and +# this line refuses the build rather than sailing through it. +Environment=MIN_MINTS_FOR_BUILD=20 +ExecStartPre=/bin/sh -c 'Q=$$(printf "\\042"); \ + n=$$(curl -sf --max-time 30 http://127.0.0.1:8788/api/mints | grep -o "$${Q}host$${Q}:" | wc -l); \ + if [ "$${n:-0}" -lt "$$MIN_MINTS_FOR_BUILD" ]; then \ + printf "<3>%s\\n" "refusing to build: /api/mints returned $${n:-0} mints, floor is $$MIN_MINTS_FOR_BUILD. Previous site left untouched."; \ + exit 1; \ + fi; \ + printf "%s\\n" "build gate: $$n mints, floor $$MIN_MINTS_FOR_BUILD"' # Check `which pnpm` on the host: a corepack or pnpm-home install sits outside /usr/bin, # and systemd's PATH does not include it. ExecStart=/usr/bin/pnpm build diff --git a/deploy/cashumints-web.service b/deploy/cashumints-web.service index efe9020..627a79b 100644 --- a/deploy/cashumints-web.service +++ b/deploy/cashumints-web.service @@ -52,6 +52,40 @@ Environment=PUBLIC_API_URL= # rather than letting the first fetch die on ECONNREFUSED. /api/health answers 503 until # it is genuinely ready, and curl -f treats that as a failure, so the loop keeps waiting. ExecStartPre=/usr/bin/timeout 90 /bin/sh -c 'until curl -sf -o /dev/null http://127.0.0.1:8788/api/health; do sleep 1; done' + +# Then: does the API actually have an index to build a site out of? +# +# Health answering 200 says the process is up and its last backfill read something. It +# does not say how many mints are in the table, and those are different questions — the +# year of ~31-event backfills had a healthy API serving a real, complete, correct list of +# eight mints. A build against that succeeds, prerenders eight cards, and rsync happily +# replaces fifty-five with eight. +# +# So count the list before spending twenty minutes building from it. Below the floor +# this exits non-zero, systemd abandons the unit at ExecStartPre, and — because publishing +# is ExecStartPost, after the build — the previously published site is never touched. The +# site stays exactly as it was and the OnFailure alert says why. +# +# Counted by the `"host":` key, one per item, rather than by counting `{`: the list +# payload carries a nested object per mint (its NUT capability switches), so brace +# counting would report roughly double. No jq: it is not installed on this host and a +# build gate should not add a dependency to run. +# +# `Q` is a double-quote character, built with printf rather than written literally, +# because this whole command is already inside systemd's single quotes and a quote of +# either kind in the grep pattern would end the argument early. +# +# A curl that fails for any reason leaves `n` empty, `$${n:-0}` reads that as zero, and +# zero is below every floor — so an API that fell over between the health check above and +# this line refuses the build rather than sailing through it. +Environment=MIN_MINTS_FOR_BUILD=20 +ExecStartPre=/bin/sh -c 'Q=$$(printf "\\042"); \ + n=$$(curl -sf --max-time 30 http://127.0.0.1:8788/api/mints | grep -o "$${Q}host$${Q}:" | wc -l); \ + if [ "$${n:-0}" -lt "$$MIN_MINTS_FOR_BUILD" ]; then \ + printf "<3>%s\\n" "refusing to build: /api/mints returned $${n:-0} mints, floor is $$MIN_MINTS_FOR_BUILD. Previous site left untouched."; \ + exit 1; \ + fi; \ + printf "%s\\n" "build gate: $$n mints, floor $$MIN_MINTS_FOR_BUILD"' # Check `which pnpm` on the host: a corepack or pnpm-home install sits outside /usr/bin, # and systemd's PATH does not include it. ExecStart=/usr/bin/pnpm build From 14548179a02c0c90b4a7afe3d2457c38240a847c Mon Sep 17 00:00:00 2001 From: michilis Date: Tue, 25 Aug 2026 16:27:33 +0200 Subject: [PATCH 5/7] Hydrate the mint lists from the live API after paint. MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit /mints was a snapshot of whatever the API held when `astro build` ran, and stayed that until the next build: a mint indexed at noon was reviewable at once — the 404 resolver saw to that — and simply had no card until 03:30. Every card's rating, review count and status were as stale as the page. The three index pages and the home page's three top-six strips now refetch `GET /api/mints?type=…` once, after paint, and rebuild their grids. The prerendered cards stay: they are the first paint, what a crawler indexes, and the whole page without JavaScript. Hydration only ever replaces them with something newer, and never with nothing — neither a failed fetch nor a well-formed empty array touches a grid that has cards in it. To make that affordable, the list payload grew the facts a chip is drawn from: `nuts`, `capabilities`, and the two probed LNURL fields. /mints and /lnurl-mints were fetching `GET /api/mints/:host` once per mint at build time to read two booleans off each; that N+1 is gone from both, which takes the build from fifty-six requests to one and is what makes the same read possible in a browser. Additive: `MintDetail` already had all four. web/src/lib/mint-cards.ts is MintCard.astro's parallel renderer, the same relationship review-cards.ts has with the reviews panel. Same classes, same data-* attributes — the sort, the search, the rank chips and the shared-element view transitions all read the DOM — and the same i18n, through the page's own inlined catalog rather than a build-time one. Base.astro gained `clientNamespaces`, so the home page can inline the `home.` catalog its strips need to rewrite "All 60 mints →" without putting 2KB of marketing copy on 1,300 mint pages. check-i18n reads the prop off the page, so the two cannot disagree. Verified in Chromium against the built site: 60 prerendered cards become 61 including a mint inserted after the build; sort, search and hide-offline operate on the new cards; /es/mints renders "En línea", "54 reseñas", "4,9" and "Solo fundir"; JavaScript disabled still shows all 60; an aborted or empty API leaves the grid alone; and a navigation away and back re-hydrates. 2016 pages build, link and hreflang checks pass, 30 web tests pass. Co-Authored-By: Claude Opus 5 --- api/src/queries.ts | 64 +++- shared/src/types.ts | 36 +++ web/scripts/check-i18n.mjs | 16 +- web/src/i18n/index.ts | 13 +- web/src/layouts/Base.astro | 12 +- web/src/lib/mint-cards.ts | 312 ++++++++++++++++++++ web/src/lib/skeleton-fixtures.ts | 3 + web/src/pages/[...locale]/fedimints.astro | 42 ++- web/src/pages/[...locale]/index.astro | 82 ++++- web/src/pages/[...locale]/lnurl-mints.astro | 82 +++-- web/src/pages/[...locale]/mints.astro | 68 ++++- 11 files changed, 670 insertions(+), 60 deletions(-) create mode 100644 web/src/lib/mint-cards.ts diff --git a/api/src/queries.ts b/api/src/queries.ts index 0d59b46..1f2f594 100644 --- a/api/src/queries.ts +++ b/api/src/queries.ts @@ -3,6 +3,7 @@ import { compareMints, NEUTRAL_PRIOR_MEAN, parseNuts, + readCapabilities, type Health, type MintDetail, type MintInfo, @@ -106,6 +107,39 @@ function round1(n: number | null): number | null { return n === null ? null : Math.round(n * 10) / 10; } +/** + * The NUT numbers a row publishes. + * + * `nuts_json` is what the prober wrote and wins; `info.nuts` is the raw NUT-06 object it + * was derived from, kept as a fallback for rows written before that column existed. One + * function so a list card and a detail page can never read a different answer off the + * same row. + */ +function rowNuts(row: MintRow, info: MintInfo | null): string[] { + if (row.nuts_json) { + try { + const parsed = JSON.parse(row.nuts_json) as string[]; + if (parsed.length > 0) return parsed; + } catch { + // Fall through to the info object below. + } + } + return info ? parseNuts(info.nuts) : []; +} + +/** + * One list item. + * + * The chip fields at the bottom are why this now parses `info_json`. The alternative was + * what /mints and /lnurl-mints used to do: fetch `GET /api/mints/:host` once per mint to + * read two booleans off each one. That is an acceptable price for a build machine + * rendering fifty-five cards once a night and an unacceptable one for every browser that + * opens the page, which is what the list has to survive now that it hydrates. + * + * Facts, not sentences. `capabilities` is two booleans and `mintChip` turns them into + * "Melt only" in the reader's language, wherever the card is being drawn. Rendering the + * label here would ship one language to twenty-four locales. + */ function toListItem(row: MintRow, agg: AggRow | undefined, mean: number, now: number): MintListItem { const base = { review_count: agg?.review_count ?? 0, @@ -114,6 +148,15 @@ function toListItem(row: MintRow, agg: AggRow | undefined, mean: number, now: nu last_review_at: agg?.last_review_at ?? null, }; + const info = parseInfo(row.info_json); + // A federation and an LNURL mint have no `info_json` and so get null, which is the + // honest value: not "both NUTs are enabled", but "there is nothing here to read". + const capabilities = info ? readCapabilities(info.nuts) : null; + + // Only the two facts the LNURL chip is drawn from, not the whole ecosystem blob: this + // payload is fetched by every visitor on three pages. + const lnurl = row.type === 'lnurl' ? parseEcosystem(row) : null; + return { url: row.url, host: row.host, @@ -127,6 +170,14 @@ function toListItem(row: MintRow, agg: AggRow | undefined, mean: number, now: nu score: bayesianScore(base, mean, now), last_review_at: base.last_review_at, version: row.version, + nuts: rowNuts(row, info), + capabilities, + ...(lnurl + ? { + max_withdrawable_msat: lnurl.max_withdrawable_msat ?? null, + funding_available: lnurl.funding_available ?? null, + } + : {}), }; } @@ -233,16 +284,9 @@ export async function getMintDetail(host: string): Promise { const item = toListItem(row, agg.get(row.url), mean, now); const info = parseInfo(row.info_json); - - let nuts: string[] = []; - if (row.nuts_json) { - try { - nuts = JSON.parse(row.nuts_json) as string[]; - } catch { - nuts = []; - } - } - if (nuts.length === 0 && info) nuts = parseNuts(info.nuts); + // `item.nuts` is the same read, through `rowNuts`. It used to be computed a second + // time here with a subtly different fallback rule; one function now answers for both. + const nuts = item.nuts; /* * Type-specific columns are spread across the payload rather than nested under a key. diff --git a/shared/src/types.ts b/shared/src/types.ts index d8691f1..9aaaa66 100644 --- a/shared/src/types.ts +++ b/shared/src/types.ts @@ -1,5 +1,7 @@ /** Shapes returned by the API. The web app builds against these. */ +import type { MintCapabilities } from './warnings.js'; + /** * Statuses a listed thing can be in. * @@ -35,6 +37,40 @@ export interface MintListItem { score: number; last_review_at: number | null; version: string | null; + + /* ---- card chips ---- + * + * The three fields below exist so a card can be drawn from the list payload alone. + * Before them, /mints fetched `GET /api/mints/:host` once per mint at build time just + * to read two booleans off each one, which is fine for fifty-five mints on one build + * machine and is not fine for every visitor's browser once the list hydrates. They are + * facts, never rendered strings: the label a chip prints is decided by `mintChip` in + * the reader's own language, on whichever side is drawing the card. + * + * A federation has no counterpart and needs none — it publishes no capability list, so + * `mintChip` returns null for one and always will. See the Fedimint branch of + * `getMintWarnings`. + */ + + /** + * NUT numbers this mint publishes, as strings: `["4", "5", "17"]`. Cashu only; empty + * for the other ecosystems and for a mint whose `/v1/info` has never been read. + */ + nuts: string[]; + /** + * NUT-04 and NUT-05 switches, the Cashu chip's only input. null means nothing is + * cached for this mint, which is a different fact from "both are on" — see + * `readCapabilities`. + */ + capabilities: MintCapabilities | null; + /** + * LNURL: the advertised withdraw ceiling, millisatoshi. Optional rather than + * `| null`, so this stays exactly what `Partial` declares on `MintDetail` + * and the two do not have to be kept identical by hand. + */ + max_withdrawable_msat?: number | null; + /** LNURL: whether the last probe reached the mint's funding node. */ + funding_available?: boolean | null; } export interface ProbeSample { diff --git a/web/scripts/check-i18n.mjs b/web/scripts/check-i18n.mjs index c4700ab..a8850bc 100644 --- a/web/scripts/check-i18n.mjs +++ b/web/scripts/check-i18n.mjs @@ -268,6 +268,20 @@ for (const file of files) { // A .ts file under lib/ or scripts/ is island code wholesale. const shipsToBrowser = /\/(lib|scripts)\//.test(relative) && relative.endsWith('.ts'); + /* + * A page can inline one extra namespace for its own islands, through Base.astro's + * `clientNamespaces` prop. The home page does: its grids hydrate from the API and + * rewrite their own "All 56 mints" links, whose strings live under `home.` — a + * namespace not worth inlining on 1,300 mint pages that never read it. + * + * Read out of the page rather than listed here, so the prop and this check cannot + * disagree. A namespace a page does not actually pass is still a leak. + */ + const extraNamespaces = new Set( + [...(/clientNamespaces=\{\[([^\]]*)\]\}/.exec(source)?.[1] ?? '').matchAll(/'([\w-]+)'/g)] + .map((m) => m[1]), + ); + for (const match of source.matchAll(T_CALL)) { const key = match[2]; used.add(key); @@ -280,7 +294,7 @@ for (const file of files) { for (const match of clientSource.matchAll(T_CALL)) { const key = match[2]; const namespace = key.split('.')[0]; - if (!clientNamespaces.has(namespace)) { + if (!clientNamespaces.has(namespace) && !extraNamespaces.has(namespace)) { clientLeaks.push({ key, file: relative, namespace }); } } diff --git a/web/src/i18n/index.ts b/web/src/i18n/index.ts index e710a94..584ec02 100644 --- a/web/src/i18n/index.ts +++ b/web/src/i18n/index.ts @@ -121,13 +121,22 @@ export function missingKeys(): Record { * key that survives is resolved: a key this locale is missing arrives already filled * with the English string, so the browser needs no fallback catalog and ships exactly * one language. + * + * `extra` is for a namespace exactly one page's islands need. The home page's grids + * hydrate and have to rewrite their own "All 56 mints →" links, which live under + * `home.` — a namespace worth about 2KB that every other page, including 1,300 mint + * pages, has no use for. Passed per page through `Base.astro`'s `clientNamespaces` + * prop, it is inlined where it is read and nowhere else. `check-i18n.mjs` does not know + * about this, so a key reached this way must still be in a namespace the checker + * accepts, or listed in `CLIENT_NAMESPACES` — see the note there. */ -export function clientCatalog(locale: Locale): Catalog { +export function clientCatalog(locale: Locale, extra: readonly string[] = []): Catalog { const catalog = catalogFor(locale); + const allowed = new Set([...CLIENT_NAMESPACES, ...extra]); const out: Catalog = {}; for (const key of Object.keys(BASE_CATALOG)) { const namespace = key.split('.')[0] ?? ''; - if (!(CLIENT_NAMESPACES as readonly string[]).includes(namespace)) continue; + if (!allowed.has(namespace)) continue; out[key] = catalog[key] ?? BASE_CATALOG[key]!; } return out; diff --git a/web/src/layouts/Base.astro b/web/src/layouts/Base.astro index d166e64..bda7e5c 100644 --- a/web/src/layouts/Base.astro +++ b/web/src/layouts/Base.astro @@ -25,6 +25,14 @@ interface Props { description: string; current?: 'mints' | 'fedimints' | 'lnurl-mints' | 'reviews' | 'wallets' | 'about'; mintCount?: number; + /** + * Catalog namespaces this page's islands need on top of `CLIENT_NAMESPACES`. + * + * The home page passes `['home']`: its three grids refresh from the API and rewrite + * their own "All 56 mints →" links, so those strings have to reach the browser. They + * are inlined on the one page that reads them rather than on all 1,300. + */ + clientNamespaces?: readonly string[]; ogType?: string; /** * A real page that should not be in the index. @@ -70,7 +78,7 @@ interface Props { const { title, description, current, mintCount, ogType = 'website', noindex = false, offGraph = false, schema = [], image = OG_IMAGE, - imageAlt, + imageAlt, clientNamespaces = [], } = Astro.props; /* @@ -118,7 +126,7 @@ const ogAlternates = LOCALES.filter((l) => l.code !== locale).map((l) => l.og); * travels in the HTML the page was sending anyway, costs no extra request, and is on * screen before the first island has finished downloading. */ -const i18nPayload = JSON.stringify({ locale, catalog: clientCatalog(locale) }).replace(/` + : `${escapeHtml(initials(name))}`; + + const statsHtml = + mint.rating_avg === null + ? `${escapeHtml(t('card.noRatings'))}` + : `` + + `${escapeHtml(f.decimal(mint.rating_avg))}` + + `` + + ``; + + /* + * "Announced" has no "last seen" to print, and printing one anyway is exactly the fake + * status this site refuses to show. What it has instead is the fact that nothing has + * checked it, said in as many words. + */ + const last = announced + ? t('card.notChecked') + : offline + ? mint.last_online + ? t('card.lastSeen', { when: f.relative(mint.last_online) }) + : t('card.neverSeen') + : mint.last_review_at + ? t('card.reviewed', { when: f.relative(mint.last_review_at) }) + : t('card.noReviews'); + + const classes = ['mint-card']; + if (offline) classes.push('is-offline'); + if (revealed) classes.push('in'); + + return ( + `` + + `
` + + iconHtml + + `` + + `${escapeHtml(name)}` + + `${escapeHtml(domain)}` + + `` + + (rank === undefined + ? '' + : `#${rank}`) + + `
` + + `
` + + statsHtml + + `${escapeHtml(t('card.reviews', { n: mint.review_count }))}` + + `
` + + `` + + `
` + + `` + + `${escapeHtml(statusLabel(mint.status, t))}` + + (chip ? `${escapeHtml(chip.label)}` : '') + + `${escapeHtml(last)}` + + `
` + + `
` + ); +} + +/** + * One ecosystem's full listing, or null. + * + * Null on anything at all going wrong, and the caller's job is then to do nothing: the + * prerendered grid is already on screen and correct as of the last build, so a failed + * refresh should be invisible rather than an error message about a list the reader can + * see. This is the same rule the reviews panel and the pulse ticker follow. + */ +export async function fetchListing(type: string): Promise { + try { + const res = await fetch(`${apiBase}/api/mints?type=${encodeURIComponent(type)}`, { + headers: { Accept: 'application/json' }, + }); + if (!res.ok) return null; + const items = (await res.json()) as MintListItem[]; + // A well-formed empty answer is still not a reason to empty a grid that has cards in + // it. An API serving nothing is the failure the build gate exists to catch, and a + // page that renders it as "no mints" would be this bug wearing a different hat. + return Array.isArray(items) && items.length > 0 ? items : null; + } catch { + return null; + } +} + +/** + * Replace a grid's cards with freshly rendered ones. + * + * Cards whose host was already on screen and revealed are rendered revealed, so the + * common case — the list is the same list, with newer numbers — is a silent swap rather + * than forty cards fading in again. Genuinely new hosts get the ordinary entrance from + * `initReveal`, which is also what reveals anything below the fold on scroll. + * + * Returns the new card elements, in DOM order, for the caller to re-apply its sort and + * filter to. + */ +export function renderMintGrid( + grid: HTMLElement, + items: MintListItem[], + options: { ranked?: boolean; revealDelayStep?: number } = {}, +): HTMLElement[] { + const { ranked = true, revealDelayStep } = options; + const t = useI18n(); + const f = formatters(t); + const { locale } = splitLocale(window.location.pathname); + + // Which hosts the reader can already see. Keyed by host rather than by index: the list + // may have grown, shrunk or reordered, and the question is per mint. + const revealed = new Set(); + for (const card of grid.querySelectorAll('[data-mint-card]')) { + if (card.classList.contains('in')) { + const host = card.getAttribute('href')?.split('/').pop(); + if (host) revealed.add(decodeURIComponent(host)); + } + } + + grid.innerHTML = items + .map((mint, i) => + mintCardHtml(mint, locale, f, { + ...(ranked ? { rank: i + 1 } : {}), + ...(revealDelayStep === undefined ? {} : { revealDelay: i * revealDelayStep }), + revealed: revealed.has(mint.host), + }), + ) + .join(''); + + initReveal(grid); + return [...grid.querySelectorAll('[data-mint-card]')]; +} + +export interface HydrateOptions { + /** `cashu`, `fedimint` or `lnurl`. */ + type: string; + /** The grid to rebuild. */ + grid: HTMLElement; + /** Keep only the first N, for the home page's top-six strips. */ + limit?: number; + ranked?: boolean; + revealDelayStep?: number; + /** Called with the new cards and the payload they were built from, on success only. */ + onReplaced?: (cards: HTMLElement[], items: MintListItem[]) => void; +} + +/** + * Fetch one ecosystem and rebuild its grid, after paint. + * + * Deliberately silent on failure — see `fetchListing`. Deliberately unconditional on + * success: the API is the newer of the two by construction, since the prerendered grid + * is a copy of what this same endpoint said at build time. + */ +export async function hydrateMintGrid(options: HydrateOptions): Promise { + const { type, grid, limit, ranked, revealDelayStep, onReplaced } = options; + + const all = await fetchListing(type); + if (!all) return; + const items = limit === undefined ? all : all.slice(0, limit); + + const cards = renderMintGrid(grid, items, { + ...(ranked === undefined ? {} : { ranked }), + ...(revealDelayStep === undefined ? {} : { revealDelayStep }), + }); + onReplaced?.(cards, all); +} diff --git a/web/src/lib/skeleton-fixtures.ts b/web/src/lib/skeleton-fixtures.ts index ac1e0ad..2d82b59 100644 --- a/web/src/lib/skeleton-fixtures.ts +++ b/web/src/lib/skeleton-fixtures.ts @@ -98,6 +98,9 @@ export const FIXTURE_MINT: MintDetail = { pubkey: '0296d0aa13b6a31cf0cd974249f4c6ed579061a4705ab9a4c1b6b1e1e4d7f6f9', info: null, nuts: ['1', '2', '3', '4', '5', '6', '7', '8', '9', '10', '11', '12'], + // Both NUTs published and neither switched off, so this fixture draws no card chip — + // which is what a representative healthy mint should look like. + capabilities: { mintDisabled: false, meltDisabled: false, mintPublished: true, meltPublished: true }, first_seen: daysAgo(420), last_probe: daysAgo(0), updated_at: daysAgo(0), diff --git a/web/src/pages/[...locale]/fedimints.astro b/web/src/pages/[...locale]/fedimints.astro index a2107f0..4ce105b 100644 --- a/web/src/pages/[...locale]/fedimints.astro +++ b/web/src/pages/[...locale]/fedimints.astro @@ -247,6 +247,7 @@ const schema = [