Compile the API instead of running its TypeScript in production.

The unit's ExecStart named src/index.ts, so every start depended on the host
having Node 22.18 or newer for native type stripping. A deploy onto a host with
Node 20 met ERR_UNKNOWN_FILE_EXTENSION, exited in under a second, and was
restarted 464 times over fifteen hours with nothing anywhere going red.

api/tsconfig.json now emits to api/dist. The source keeps its explicit .ts import
specifiers, which is what makes `node --watch src/index.ts` work in development;
rewriteRelativeImportExtensions turns them into .js on the way out, so what runs
in production is ordinary ESM that any Node from 20.18 up will start.

`pnpm build` builds shared, then api, then web. `pnpm dev` is unchanged.

deploy/ is tracked rather than ignored: the unit files are the thing an operator
copies to /etc/systemd/system, and the alert unit added next has to live
somewhere a deploy can find it.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
michilis
2026-08-25 15:58:53 +02:00
co-authored by Claude Opus 5
parent 06ba3d35e7
commit 65307ba278
10 changed files with 455 additions and 17 deletions
+61
View File
@@ -0,0 +1,61 @@
# /etc/systemd/system/cashumints-site.service
#
# Serves the built site on loopback. nginx proxies to it and never opens a file itself,
# which is the point: when nginx held a `root` inside /home/cashumints, every directory
# down to dist had to be traversable by www-data, and the one that was not took the
# whole site down as a blanket 404 with nothing in the error log naming the cause.
#
# This is a long-running daemon, unlike cashumints-web.service next to it — that one is
# the oneshot that produces what this one serves.
[Unit]
Description=cashumints.space static site server
Wants=network-online.target
After=network-online.target
# Give up after five failures in a minute instead of restarting forever. A process that
# cannot start will not start on the 4000th attempt either, and `failed` in
# `systemctl status` is a far louder signal than a journal scrolling past. These two are
# [Unit] keys; systemd ignores them under [Service] with only a warning.
StartLimitIntervalSec=60
StartLimitBurst=5
# Not Requires=cashumints.service: the pages are prerendered, so the site keeps serving
# a correct-as-of-last-build copy while the API is down. Only the islands go quiet.
[Service]
Type=simple
User=cashumints
Group=cashumints
WorkingDirectory=/home/cashumints/CashuMints.space/web
# The tree comes from cashumints-web.service, which rsyncs it here after a build.
# Serving web/dist directly would mean a rebuild empties the site for the length of it.
StateDirectory=cashumints
Environment=NODE_ENV=production
Environment=SITE_PORT=8789
Environment=SITE_HOST=127.0.0.1
Environment=WEB_ROOT=/var/lib/cashumints/web
ExecStart=/usr/bin/node server.mjs
Restart=on-failure
RestartSec=5s
KillSignal=SIGTERM
# In-flight responses finish; idle keep-alive connections are closed at once.
TimeoutStopSec=15s
UMask=0027
NoNewPrivileges=true
PrivateTmp=true
PrivateDevices=true
ProtectSystem=strict
# Read-only rather than absent: server.mjs itself lives under /home/cashumints.
ProtectHome=read-only
ReadWritePaths=/var/lib/cashumints
ProtectKernelTunables=true
ProtectKernelModules=true
ProtectControlGroups=true
RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6
RestrictSUIDSGID=true
LockPersonality=true
[Install]
WantedBy=multi-user.target
+90
View File
@@ -0,0 +1,90 @@
# /etc/systemd/system/cashumints-web.service
#
# The frontend is static: `output: 'static'` in astro.config.mjs, and the whole site is
# produced ahead of time. There is no frontend build to keep alive, so this unit is a
# build rather than a daemon — one shot of `pnpm build`, which compiles shared/, renders
# a social card per mint and prerenders every page from the live API. The daemon that
# hands the result out is cashumints-site.service.
#
# Run it after a deploy:
# sudo systemctl start cashumints-web
#
# Mint pages are prerendered, so new mints and new review counts only appear at the
# next build; pair this with a .timer for the nightly rebuild. There is deliberately no
# [Install] section — a rebuild should be scheduled, not fired on every boot.
[Unit]
Description=Rebuild the cashumints.space static site
# Every page's data comes from the API over loopback, so the API has to be up.
# Requires= rather than Wants=: a dead API should abort the build, not replace a good
# site with an empty one.
Requires=cashumints.service
After=cashumints.service network-online.target
Wants=network-online.target
[Service]
Type=oneshot
User=cashumints
Group=cashumints
WorkingDirectory=/home/cashumints/CashuMints.space
# Where the published copy lands. Shared with the API and the site server, and created
# by systemd with this unit's ownership if it is not there yet.
StateDirectory=cashumints
Environment=NODE_ENV=production
# Where the build reaches the API. Must match PORT= in cashumints.service.
Environment=API_URL=http://127.0.0.1:8788
Environment=SITE_URL=https://cashumints.space
# Browser-facing origin. Empty means same origin: islands fetch /api/... and nginx
# forwards it. Set this only if the API ever moves to its own hostname. Declared here
# even though it is empty, because systemd's environment wins over .env — so what a
# production build emits cannot drift with an edit to that file.
Environment=PUBLIC_API_URL=
# After= orders the start; it does not wait for the port to accept connections. At boot
# the API is still opening its database and probing, so block until it reports healthy
# rather than letting the first fetch die on ECONNREFUSED. /api/health answers 503 until
# it is genuinely ready, and curl -f treats that as a failure, so the loop keeps waiting.
ExecStartPre=/usr/bin/timeout 90 /bin/sh -c 'until curl -sf -o /dev/null http://127.0.0.1:8788/api/health; do sleep 1; done'
# Check `which pnpm` on the host: a corepack or pnpm-home install sits outside /usr/bin,
# and systemd's PATH does not include it.
ExecStart=/usr/bin/pnpm build
# Publish, as a separate step from building.
#
# `astro build` empties dist before it writes, so the site server cannot read dist
# directly — a nightly rebuild would be a nightly minute of 404s. It serves this copy
# instead, and the copy is only touched once a build has succeeded: a failed build
# leaves the previous site up rather than replacing it with a half-written one, which is
# the same reason Requires=cashumints.service is above.
#
# --delay-updates stages the changed files and renames them in at the end, so the window
# where the tree is a mix of two builds is a rename rather than a whole transfer, and
# --delete-after keeps removals from landing before their replacements. Unchanged files
# — every hashed asset and card, which is nearly all of it — are not touched at all.
ExecStartPost=/usr/bin/rsync -a --delete-after --delay-updates web/dist/ /var/lib/cashumints/web/
# ~200 prerendered pages plus a card per mint. Minutes, not seconds, on a small VPS, and
# TimeoutStartSec is what bounds a Type=oneshot.
TimeoutStartSec=1800
# A nightly rebuild should not starve the API it is reading from.
Nice=10
# The site server runs as cashumints and reads its own files, so this no longer has to
# be world-readable — it was 0022 for nginx, back when nginx opened the files as
# www-data. Kept at 0022 anyway: rsync preserves these modes into the published copy,
# and a readable static site is easier to inspect than one that needs sudo.
UMask=0022
NoNewPrivileges=true
PrivateTmp=true
PrivateDevices=true
# ProtectHome is deliberately absent, unlike in cashumints.service: this unit writes
# inside /home/cashumints — web/dist, web/public/og, web/src/generated and the pnpm
# store are all under it.
ProtectSystem=full
ProtectKernelTunables=true
ProtectKernelModules=true
ProtectControlGroups=true
RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6
+25
View File
@@ -0,0 +1,25 @@
# /etc/systemd/system/cashumints-web.timer
#
# The nightly rebuild. Mint pages are prerendered, so a new mint or a new review count
# appears at the next build; between builds the site stays correct because the islands
# refresh status and reviews at runtime, and a mint indexed since the last build still
# resolves through the client-side fallback on the 404 page.
#
# sudo systemctl enable --now cashumints-web.timer
#
# The service it starts builds first and publishes second, so a failed build leaves the
# running site untouched rather than replacing it with a half-written one.
[Unit]
Description=Nightly cashumints.space rebuild
[Timer]
OnCalendar=*-*-* 03:30:00
# Run it on the next boot if the machine was off at 03:30, rather than skipping a day.
Persistent=true
# Without this every host rebuilds on the same second. Harmless with one VPS; free
# insurance if there is ever a second.
RandomizedDelaySec=15m
[Install]
WantedBy=timers.target
+56
View File
@@ -0,0 +1,56 @@
# /etc/systemd/system/cashumints.service
[Unit]
Description=cashumints.space indexer and API
Wants=network-online.target
After=network-online.target
# Stop after five failures in a minute rather than restarting forever. A wrong Node on
# PATH once produced four thousand identical crashes in the journal before anyone read
# one of them; `failed` in `systemctl status` says the same thing in one line. Both keys
# belong to [Unit] — under [Service] systemd only warns and ignores them.
StartLimitIntervalSec=60
StartLimitBurst=5
[Service]
Type=simple
User=cashumints
Group=cashumints
WorkingDirectory=/home/cashumints/CashuMints.space/api
# StateDirectory creates /var/lib/cashumints with the service user's ownership.
StateDirectory=cashumints
Environment=NODE_ENV=production
Environment=PORT=8788
Environment=DB_PATH=/var/lib/cashumints/cashumints.db
Environment=ICON_DIR=/var/lib/cashumints/icons
# Compiled JavaScript, run by the distribution's own node.
#
# This line used to read `src/index.ts`, which made every start depend on the host
# having Node 22.18 or newer for native type stripping. A host with Node 20 answered
# that with ERR_UNKNOWN_FILE_EXTENSION in under a second, 464 times over fifteen hours,
# and nothing anywhere went red. `pnpm build` now emits api/dist, so what runs here is
# ordinary ESM and any Node from 20.18 up will start it.
#
# Deliberately /usr/bin/node and nothing else: an nvm or fnm path is invisible to this
# unit's ProtectHome and breaks silently at the next version bump.
ExecStart=/usr/bin/node --env-file-if-exists=../.env dist/index.js
Restart=on-failure
RestartSec=5s
KillSignal=SIGTERM
TimeoutStopSec=30s
UMask=0027
NoNewPrivileges=true
PrivateTmp=true
ProtectSystem=strict
ProtectHome=read-only
ReadWritePaths=/var/lib/cashumints
PrivateDevices=true
ProtectKernelTunables=true
ProtectKernelModules=true
ProtectControlGroups=true
RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6
[Install]
WantedBy=multi-user.target
+160
View File
@@ -0,0 +1,160 @@
# /etc/nginx/sites-available/cashumints.space
#
# nginx terminates TLS and proxies. It opens no file belonging to this project — not the
# built site, not an icon — and that is deliberate.
#
# It used to point a `root` at web/dist. Because nginx runs as www-data and everything
# this project owns runs as cashumints, that arrangement required every directory from /
# down to dist to be traversable by a user with no other business in the tree. A home
# directory at its default 0700 anywhere in that chain broke the entire site, and it
# broke it invisibly: `try_files` treats a permission error as a plain miss, so the
# symptom was a blanket 404, or an internal-redirect loop that ended in a 500 with the
# real cause named nowhere.
#
# Two upstreams now, both on loopback, both owned by the same user that built what they
# serve:
#
# 127.0.0.1:8789 cashumints-site.service the prerendered site
# 127.0.0.1:8788 cashumints.service /api/* and /icons/*
#
# Routing that used to live here lives with the thing that owns it. The locale 404 rule
# in particular was a hand-maintained alternation of 23 codes in a file that is not in
# the repository; adding a language meant remembering to edit it, and forgetting was
# silent. web/server.mjs resolves those from the built tree.
proxy_cache_path /var/cache/nginx/cashumints
levels=1:2
keys_zone=cashumints:10m
max_size=256m
inactive=10m
use_temp_path=off;
# Keep a few connections open to each upstream rather than reconnecting per request.
# Both processes hold idle sockets longer than nginx does, so nginx is always the side
# that closes and there is no window where it reuses a socket the upstream just dropped
# — that race is what produces sporadic 502s under load.
upstream cashumints_site {
server 127.0.0.1:8789;
keepalive 16;
}
upstream cashumints_api {
server 127.0.0.1:8788;
keepalive 8;
}
server {
listen 80;
listen [::]:80;
server_name cashumints.space;
location /.well-known/acme-challenge/ {
root /var/www/html;
}
location / {
return 301 https://cashumints.space$request_uri;
}
}
server {
listen 443 ssl http2;
listen [::]:443 ssl http2;
server_name cashumints.space;
# nginx 1.25 and later want `http2 on;` on its own line and warn about the form above.
# Left as is because it is the form that works on both, and Debian 12 ships 1.22.
ssl_certificate /etc/letsencrypt/live/cashumints.space/fullchain.pem;
ssl_certificate_key /etc/letsencrypt/live/cashumints.space/privkey.pem;
include /etc/letsencrypt/options-ssl-nginx.conf;
ssl_dhparam /etc/letsencrypt/ssl-dhparams.pem;
# Set here rather than upstream: this is the only part of the stack that knows a
# request arrived over TLS. Add `preload` only once you are content never to serve
# this name over plain HTTP again.
add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always;
# The upstreams send no Content-Encoding, so compression is nginx's to do. gzip_proxied
# any is required — without it nginx refuses to compress a proxied response at all.
gzip on;
gzip_proxied any;
gzip_vary on;
gzip_comp_level 5;
gzip_min_length 1024;
gzip_types text/plain text/css text/javascript application/javascript application/json
application/manifest+json application/xml image/svg+xml;
# Nothing here accepts an upload. The API's largest body is an 8 KB JSON submission,
# and rejecting the oversized ones at the edge keeps them off the Node process.
client_max_body_size 16k;
# Both upstreams are a process on this machine. A slow response is a bug, not a
# network condition, and failing fast beats holding a worker for a minute.
proxy_connect_timeout 2s;
proxy_read_timeout 30s;
proxy_send_timeout 30s;
# HTTP/1.1 with an empty Connection header is what makes the keepalive pools above
# work; the default 1.0 opens a new socket per request.
proxy_http_version 1.1;
proxy_set_header Connection "";
# Every proxied location includes Debian's /etc/nginx/proxy_params, which sets Host,
# X-Real-IP, X-Forwarded-For and X-Forwarded-Proto. That include is load-bearing, not
# decorative: the API's rate limiter reads the last hop of X-Forwarded-For to tell two
# visitors apart, and without it every request arrives from the loopback peer and
# shares one bucket. On a distro that ships no proxy_params, set those four by hand.
# Do not also set X-Forwarded-For alongside the include — declaring it twice is what
# produces nginx's "could not build optimal proxy_headers_hash" warning.
# The prerendered site. Cache-Control comes from server.mjs — a year and immutable for
# anything with a content hash in its name, revalidate-every-time for markup — so
# there is nothing to restate here.
location / {
proxy_pass http://cashumints_site;
include proxy_params;
}
# Health must always reflect the live process.
location = /api/health {
proxy_pass http://cashumints_api;
proxy_cache off;
# add_header replaces rather than merges: declaring one here drops every add_header
# inherited from the server block, so HSTS has to be restated alongside it.
add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always;
add_header Cache-Control "no-store" always;
include proxy_params;
}
# Briefly cache the read-heavy endpoints.
location ~ ^/api/(mints|stats)(?:/|$|\?) {
proxy_pass http://cashumints_api;
proxy_cache cashumints;
proxy_cache_valid 200 30s;
proxy_cache_valid 404 10s;
proxy_cache_use_stale updating error timeout http_500 http_502 http_503;
proxy_cache_background_update on;
proxy_cache_lock on;
# Restated for the same reason as in /api/health above.
add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always;
add_header X-Cache-Status $upstream_cache_status always;
include proxy_params;
}
location /api/ {
proxy_pass http://cashumints_api;
include proxy_params;
}
location /icons/ {
proxy_pass http://cashumints_api;
proxy_cache cashumints;
proxy_cache_valid 200 1d;
include proxy_params;
}
# No error_page and no try_files. A miss is the site server's 404 page, in the right
# language and with a 404 status; an nginx error page here would replace it with a
# blank one and hide which upstream failed.
}