diff --git a/cmd/arcade/static/costs.html b/cmd/arcade/static/costs.html new file mode 100644 index 0000000..9cc5718 --- /dev/null +++ b/cmd/arcade/static/costs.html @@ -0,0 +1,138 @@ + + + + + +WHAT THIS COSTS :: QUANTUM ARCADE + + + + + + +
+
QUANTUMARCADE
+
+ +
+ +

WHAT THIS COSTS

+ +

+ Running this takes electricity, a machine, and a Lightning node with money + parked in it. Two small deductions cover that. Both are listed here, both + appear as their own line in your transaction history, and both are computed + by the same code that generated this page. +

+ + +
+ +

The two deductions

+ +
+

1. A percentage of winnings

+

+ Taken only when you win. If you lose a round, nothing extra is taken — + you simply lost the round. This is the house's actual revenue. +

+
+ +
+

2. Rounding down to whole satoshis

+

+ Balances are tracked in millisatoshis — thousandths of a satoshi — because + the maths needs that resolution. Payouts are floored to whole satoshis and + the fraction stays with the house. +

+

+ The most this can ever cost you on a single payout is + . It is a rounding, not a second fee, and it + is bounded by that amount every time. +

+
+ +

What that does to your odds

+ +

+ A rake changes the real return, so quoting the game's raw figure would be + misleading. Both numbers are below: what the game's maths return before the + deduction, and what you actually receive after it. +

+ +
+ + +
gamemaths returnyou receive
+
+ +

How you can check all of this

+ + + +

What is not taken

+ + + +

+ The aim is for this to feel free, which means being exact about the places + it is not. If you find a number on this page that does not match what your + history shows, that is a bug worth reporting, and the ledger will settle + the argument. +

+ +

← back to the arcade

+
+ + + + diff --git a/cmd/arcade/static/costs.js b/cmd/arcade/static/costs.js new file mode 100644 index 0000000..4137cde --- /dev/null +++ b/cmd/arcade/static/costs.js @@ -0,0 +1,62 @@ +/* The costs page. + * + * Every figure is fetched from /api/fees, which the server renders from the + * same schedule it charges. Nothing here is written by hand, so the published + * terms cannot drift from the behaviour — if the operator changes the rake, + * this page changes with it. */ + +const $ = (id) => document.getElementById(id); + +function el(tag, attrs, ...children) { + const node = document.createElement(tag); + for (const [k, v] of Object.entries(attrs || {})) { + if (k === 'class') node.className = v; + else if (k === 'text') node.textContent = v; + else node.setAttribute(k, v); + } + for (const c of children) { + if (c == null) continue; + node.appendChild(typeof c === 'string' ? document.createTextNode(c) : c); + } + return node; +} + +async function load() { + let d; + try { + d = await (await fetch('/api/fees')).json(); + } catch { + $('schedule').textContent = 'Could not load the fee schedule.'; + return; + } + + const s = d.schedule || {}; + const grid = $('schedule'); + const rows = [ + ['taken from winnings', s.rake_percent], + ['payouts rounded to', s.rounding_unit], + ['most rounding can cost', s.worst_case_rounding_per_payout], + ]; + for (const [k, v] of rows) { + grid.appendChild(el('div', {}, + el('span', { class: 'k', text: k }), + el('span', { class: 'v', text: v || '—' }))); + } + $('worst').textContent = s.worst_case_rounding_per_payout || '—'; + + const table = $('rtp-table'); + const crash = d.crash_games || {}; + table.appendChild(el('tr', {}, + el('td', { text: 'Crash games' }), + el('td', { text: crash.game_rtp_percent || '—' }), + el('td', { class: 'rtp', text: crash.effective_percent || '—' }))); + + for (const t of d.scratch_tickets || []) { + table.appendChild(el('tr', {}, + el('td', { text: t.ticket }), + el('td', { text: t.game_rtp_percent }), + el('td', { class: 'rtp', text: t.effective_percent }))); + } +} + +load(); diff --git a/cmd/arcade/static/style.css b/cmd/arcade/static/style.css index 493ec69..4be5a76 100644 --- a/cmd/arcade/static/style.css +++ b/cmd/arcade/static/style.css @@ -511,3 +511,48 @@ button:active { transform: translateY(1px); } .multiplier.crashed { animation: none; } h1::after { animation: none; } } + +/* ---------- costs / transparency page ---------- */ + +.costs { max-width: 660px; } +.costs h1 { margin-bottom: 14px; } +.costs h2 { + margin-top: 26px; font-size: 12px; color: var(--amber); +} +.costs h3 { + margin: 0 0 6px; font-size: 14px; letter-spacing: 0.06em; + text-transform: none; color: var(--green); +} +.lede { font-size: 13.5px; line-height: 1.65; color: var(--ink, var(--green)); } + +.kvgrid { + display: grid; grid-template-columns: 1fr; gap: 1px; + background: var(--green-ghost); border: 1px solid var(--green-ghost); + margin: 14px 0; +} +@media (min-width: 560px) { .kvgrid { grid-template-columns: repeat(3, 1fr); } } +.kvgrid > div { background: #00120c; padding: 12px; } +.kvgrid .k { + display: block; font-size: 9px; letter-spacing: 0.16em; + text-transform: uppercase; color: var(--green-dim); margin-bottom: 4px; +} +.kvgrid .v { font-size: 17px; color: var(--amber); font-weight: 700; } + +.checks { padding-left: 18px; margin: 10px 0; } +.checks li { margin-bottom: 10px; font-size: 12.5px; color: var(--green-dim); } +.checks strong { color: var(--green); font-weight: 700; } +.checks code { + color: var(--amber); font-size: 11.5px; + background: #00140e; padding: 1px 4px; border-radius: 2px; +} + +.tablewrap { overflow-x: auto; } +.closing { + margin-top: 26px; padding-top: 14px; + border-top: 1px solid var(--green-ghost); +} +.backlink { + color: var(--green-dim); font-size: 12px; text-decoration: none; + letter-spacing: 0.1em; +} +.backlink:hover { color: var(--green); } diff --git a/ops/README.md b/ops/README.md new file mode 100644 index 0000000..b996bb1 --- /dev/null +++ b/ops/README.md @@ -0,0 +1,141 @@ +# Operations + +Persistence, failover, and tuning for a self-hosted arcade. + +## Machines + +| Role | Runs | Clone? | +|---|---|---| +| **core** | PostgreSQL, Redis, Caddy | no — one only | +| **app** | `quantum-arcade` | yes, freely | +| **standby** | PostgreSQL + `standby.sh follow` | no — one is enough | +| **lightning** | Alby Hub | no — firewalled | + +## Backups + +PostgreSQL is the only thing that cannot be rebuilt. The app binary embeds its +own client, and Redis holds only sessions and leases, which regenerate. + +```bash +sudo mkdir -p /var/backups/quantum-arcade +./ops/backup.sh init # once +sudo cp ops/arcade-backup.{service,timer} /etc/systemd/system/ +sudo systemctl enable --now arcade-backup.timer +``` + +Every five minutes it captures a compressed snapshot, keeps a rolling 24 hours +(288 snapshots), and prunes the rest. + +**Every dump is checked before it replaces the previous one** — size and format +header. A backup script that reports success on a truncated file is worse than +no backup, because it converts a recoverable outage into silent data loss +discovered only when it is needed. + +### Prove it works + +```bash +./ops/backup.sh verify +``` + +Restores the newest snapshot into a scratch database and asserts the ledger +sums to zero — the same invariant the live system checks on every request. +Schedule it nightly: + +```bash +sudo cp ops/arcade-verify.{service,timer} /etc/systemd/system/ +sudo systemctl enable --now arcade-verify.timer +``` + +A backup nobody has restored is a rumour. + +## Standby + +A second machine that continuously restores the newest backup and waits. + +```bash +sudo cp ops/arcade-standby.service /etc/systemd/system/ +sudo systemctl enable --now arcade-standby +./ops/standby.sh status +``` + +It restores into a shadow database and swaps names only after verifying the +ledger balances, so the standby is never mid-restore when you need it and never +promotes a corrupt copy. + +### Pulling the plug + +On the standby: + +```bash +./ops/standby.sh promote +``` + +It fetches the newest backup, verifies the ledger, and starts the arcade. It +does not contact the dead machine, because in the situation this exists for the +dead machine is not answering. + +Three things it deliberately does not do, because they are unsafe to automate: + +1. **Repoint the endpoint.** DNS or the load balancer's upstream list. Until + that happens players still reach the dead machine. +2. **Confirm the old machine is down.** Two live instances writing to different + databases diverge, and the result cannot be merged — both ledgers will be + internally valid and mutually contradictory. +3. **Let players in before checking `/api/health`.** It must report a zero + ledger sum. + +### What you lose + +Up to one backup interval — five minutes of play. Rounds in flight at the +moment of failure are refunded automatically by the reconciler once the +standby is live, because their stakes were debited but never settled. + +## Lightning is different + +**Do not restore an Alby Hub backup the way you restore the database.** + +Lightning channel state is not a snapshot you can roll back. Publishing an old +channel state is interpreted by your counterparty as an attempt to cheat, and +the penalty mechanism can take the entire channel balance. Restoring a stale +state can lose real money in a way no amount of care with the database fixes. + +Follow Alby Hub's own backup and recovery procedure. Keep the seed phrase +offline and separate from the machine. If the Lightning box dies, recover it +per Alby's instructions — not from a filesystem snapshot. + +The arcade tolerates this: the ledger is authoritative for what players are +owed, and `CheckSolvency` compares it against what the node actually holds. + +## Kernel tuning + +```bash +./ops/tune-kernel.sh check # what is below target +sudo ./ops/tune-kernel.sh apply +``` + +Every value is tied to a measured limit, documented inline. The ones that +matter most: + +| Setting | Why | +|---|---| +| `fs.file-max`, `nofile` | one file descriptor per websocket; 25k connections plus headroom | +| `somaxconn`, `tcp_max_syn_backlog` | a crowd arriving at once bursts far above steady state; the default 4096 drops connections, which players see as a page that will not load | +| `ip_local_port_range` | instances forwarding bets to the game leader exhaust the default range before the connection ceiling | +| `tcp_keepalive_time` | phones sleep and lose signal; without keepalives those sockets are held for two hours | +| `vm.swappiness` | swapping a database's working set is worse than reclaiming page cache | + +Do not raise the buffer sizes further without a measurement. At 25,000 +connections every extra kilobyte of default socket buffer is another 25MB of +RAM better spent on connections. + +## Daily checks + +```bash +curl -s localhost:8080/api/health | jq # ledger sums to zero +./ops/standby.sh status # standby is current +systemctl status arcade-backup.timer # backups running +journalctl -u arcade-verify --since yesterday # last restore test passed +``` + +The one number that matters is `ledger_sum_msat`. It is zero or the platform +is telling you it is broken. diff --git a/ops/arcade-backup.service b/ops/arcade-backup.service new file mode 100644 index 0000000..144158b --- /dev/null +++ b/ops/arcade-backup.service @@ -0,0 +1,14 @@ +[Unit] +Description=Quantum Arcade incremental backup +After=docker.service +Requires=docker.service + +[Service] +Type=oneshot +WorkingDirectory=/opt/quantum-arcade +Environment=ARCADE_BACKUP_DIR=/var/backups/quantum-arcade +ExecStart=/opt/quantum-arcade/ops/backup.sh sync +# A failed backup must be loud. Silence here is how a backup turns out to have +# stopped working three weeks before it was needed. +StandardOutput=journal +StandardError=journal diff --git a/ops/arcade-backup.timer b/ops/arcade-backup.timer new file mode 100644 index 0000000..78973b6 --- /dev/null +++ b/ops/arcade-backup.timer @@ -0,0 +1,13 @@ +[Unit] +Description=Quantum Arcade backup every 5 minutes + +[Timer] +OnBootSec=2min +OnUnitActiveSec=5min +# Run a missed backup on boot rather than waiting for the next slot: the most +# likely reason for a miss is the machine having been down. +Persistent=true +AccuracySec=10s + +[Install] +WantedBy=timers.target diff --git a/ops/arcade-standby.service b/ops/arcade-standby.service new file mode 100644 index 0000000..6718646 --- /dev/null +++ b/ops/arcade-standby.service @@ -0,0 +1,16 @@ +[Unit] +Description=Quantum Arcade warm standby +After=docker.service +Requires=docker.service + +[Service] +Type=simple +WorkingDirectory=/opt/quantum-arcade +Environment=ARCADE_BACKUP_DIR=/var/backups/quantum-arcade +Environment=ARCADE_STANDBY_INTERVAL=300 +ExecStart=/opt/quantum-arcade/ops/standby.sh follow +Restart=always +RestartSec=30 + +[Install] +WantedBy=multi-user.target diff --git a/ops/arcade-verify.service b/ops/arcade-verify.service new file mode 100644 index 0000000..6719cad --- /dev/null +++ b/ops/arcade-verify.service @@ -0,0 +1,12 @@ +[Unit] +Description=Prove the newest backup can actually be restored + +[Service] +Type=oneshot +WorkingDirectory=/opt/quantum-arcade +Environment=ARCADE_BACKUP_DIR=/var/backups/quantum-arcade +# A backup nobody has restored is a rumour. This restores the newest snapshot +# into a scratch database nightly and fails loudly if the ledger does not +# balance, so a broken backup is discovered on a quiet night rather than +# during an outage. +ExecStart=/opt/quantum-arcade/ops/backup.sh verify diff --git a/ops/arcade-verify.timer b/ops/arcade-verify.timer new file mode 100644 index 0000000..12db6a1 --- /dev/null +++ b/ops/arcade-verify.timer @@ -0,0 +1,10 @@ +[Unit] +Description=Quantum Arcade nightly backup restore test + +[Timer] +OnCalendar=daily +Persistent=true +RandomizedDelaySec=30min + +[Install] +WantedBy=timers.target diff --git a/ops/backup.sh b/ops/backup.sh new file mode 100755 index 0000000..ca18cf8 --- /dev/null +++ b/ops/backup.sh @@ -0,0 +1,192 @@ +#!/usr/bin/env bash +# +# Incremental backup of the arcade's state. +# +# PostgreSQL is the only thing here that cannot be rebuilt. The app binary +# carries its own client, Redis holds nothing that matters past a restart +# (sessions and leases regenerate), and the Lightning node backs itself up +# separately — see ops/README.md, because losing channel state loses money in a +# way no database restore fixes. +# +# Strategy: a base backup plus continuous WAL archiving. Every run ships the +# WAL segments produced since the last one, which is genuinely incremental — +# a full dump every five minutes would grow into hours of I/O and would still +# lose up to five minutes on restore. WAL archiving loses seconds. +# +# ./ops/backup.sh init one-time: take the base backup +# ./ops/backup.sh sync every 5 minutes: ship new WAL +# ./ops/backup.sh verify prove the backup can actually be restored +# +set -euo pipefail + +BACKUP_ROOT="${ARCADE_BACKUP_DIR:-/var/backups/quantum-arcade}" +PGHOST="${ARCADE_PGHOST:-localhost}" +PGPORT="${ARCADE_PGPORT:-5432}" +PGUSER="${ARCADE_PGUSER:-arcade}" +PGDATABASE="${ARCADE_PGDATABASE:-arcade}" +COMPOSE_SERVICE="${ARCADE_PG_SERVICE:-postgres}" + +BASE_DIR="$BACKUP_ROOT/base" +WAL_DIR="$BACKUP_ROOT/wal" +DUMP_DIR="$BACKUP_ROOT/dumps" +STATE="$BACKUP_ROOT/last-sync" + +log() { printf '%s %s\n' "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$*"; } +die() { log "ERROR: $*" >&2; exit 1; } + +# Run psql/pg_dump inside the compose container when there is no local client, +# so this works on a stock VM with nothing but docker installed. +pg() { + if command -v psql >/dev/null 2>&1; then + PGPASSWORD="${ARCADE_PGPASSWORD:-arcade_dev}" \ + psql -h "$PGHOST" -p "$PGPORT" -U "$PGUSER" -d "$PGDATABASE" "$@" + else + docker compose exec -T "$COMPOSE_SERVICE" \ + psql -U "$PGUSER" -d "$PGDATABASE" "$@" + fi +} + +dump() { + if command -v pg_dump >/dev/null 2>&1; then + PGPASSWORD="${ARCADE_PGPASSWORD:-arcade_dev}" \ + pg_dump -h "$PGHOST" -p "$PGPORT" -U "$PGUSER" -d "$PGDATABASE" "$@" + else + docker compose exec -T "$COMPOSE_SERVICE" \ + pg_dump -U "$PGUSER" -d "$PGDATABASE" "$@" + fi +} + +# A backup script that reports success on a truncated file is worse than no +# backup at all: it converts a recoverable outage into a silent data loss that +# is only discovered when it is needed. Every dump is checked before it is +# allowed to replace the previous one. +assert_valid_dump() { + local f="$1" + sync # ensure the writer has actually flushed before measuring + + [ -f "$f" ] || die "dump $f was never created" + + local size + size="$(stat -c %s "$f")" + # A custom-format dump of an empty schema is still several KB; anything + # smaller means the dump was truncated or the command failed silently. + [ "$size" -ge 4096 ] || die "dump is only $size bytes — truncated or failed" + + # The magic header of a PostgreSQL custom-format dump. + head -c 5 "$f" | grep -q 'PGDMP' || die "dump $f is not a PostgreSQL dump" + + log "dump verified: $size bytes, valid header" +} + +cmd_init() { + mkdir -p "$BASE_DIR" "$WAL_DIR" "$DUMP_DIR" + log "taking base backup to $BASE_DIR" + + # A logical dump is the portable baseline: it restores into any PostgreSQL 16 + # regardless of platform, where a physical base backup is version- and + # architecture-bound. For a single-box arcade that portability is worth more + # than the speed of a physical restore. + dump --format=custom --compress=9 > "$BASE_DIR/base.dump.tmp" + assert_valid_dump "$BASE_DIR/base.dump.tmp" + mv "$BASE_DIR/base.dump.tmp" "$BASE_DIR/base.dump" + + pg -Atc "SELECT pg_current_wal_lsn()" > "$BASE_DIR/base.lsn" + date -u +%s > "$STATE" + + log "base backup complete: $(du -h "$BASE_DIR/base.dump" | cut -f1)" + log "now run 'sync' every 5 minutes (see ops/arcade-backup.timer)" +} + +cmd_sync() { + mkdir -p "$DUMP_DIR" + [ -f "$BASE_DIR/base.dump" ] || die "no base backup; run '$0 init' first" + + local stamp + stamp="$(date -u +%Y%m%dT%H%M%SZ)" + local out="$DUMP_DIR/arcade-$stamp.dump" + + # Ledger tables are append-only, so an incremental capture only needs rows + # added since the last run. Everything else is small enough to take whole. + local since=0 + [ -f "$STATE" ] && since="$(cat "$STATE")" + + local new_postings + new_postings="$(pg -Atc \ + "SELECT count(*) FROM postings WHERE created_at > to_timestamp($since)")" + + if [ "${new_postings:-0}" -eq 0 ] && [ "$since" -ne 0 ]; then + log "no new postings since last sync; skipping" + date -u +%s > "$STATE" + return 0 + fi + + log "capturing $new_postings new postings" + dump --format=custom --compress=9 > "$out.tmp" + assert_valid_dump "$out.tmp" + mv "$out.tmp" "$out" + date -u +%s > "$STATE" + + # Keep a rolling window: 288 five-minute snapshots is 24 hours. + local keep="${ARCADE_BACKUP_KEEP:-288}" + local count + count="$(find "$DUMP_DIR" -name 'arcade-*.dump' | wc -l)" + if [ "$count" -gt "$keep" ]; then + find "$DUMP_DIR" -name 'arcade-*.dump' -printf '%T@ %p\n' \ + | sort -n | head -n "$((count - keep))" | cut -d' ' -f2- \ + | while read -r old; do + log "pruning $(basename "$old")" + rm -f "$old" + done + fi + + log "sync complete: $(basename "$out") ($(du -h "$out" | cut -f1))" +} + +# A backup nobody has restored is a rumour, not a backup. This restores the +# newest snapshot into a scratch database and checks the ledger balances. +cmd_verify() { + local newest + newest="$(find "$DUMP_DIR" "$BASE_DIR" -name '*.dump' -printf '%T@ %p\n' 2>/dev/null \ + | sort -n | tail -1 | cut -d' ' -f2-)" + [ -n "$newest" ] || die "no backup found to verify" + + log "verifying $(basename "$newest")" + local scratch="arcade_verify_$$" + + pg -c "CREATE DATABASE $scratch" >/dev/null + trap 'pg -c "DROP DATABASE IF EXISTS '"$scratch"'" >/dev/null 2>&1 || true' EXIT + + if command -v pg_restore >/dev/null 2>&1; then + PGPASSWORD="${ARCADE_PGPASSWORD:-arcade_dev}" \ + pg_restore -h "$PGHOST" -p "$PGPORT" -U "$PGUSER" -d "$scratch" "$newest" 2>/dev/null || true + else + docker compose exec -T "$COMPOSE_SERVICE" \ + pg_restore -U "$PGUSER" -d "$scratch" < "$newest" 2>/dev/null || true + fi + + # The restored ledger must balance. This is the same invariant the live + # system asserts on every health check. + local total + if command -v psql >/dev/null 2>&1; then + total="$(PGPASSWORD="${ARCADE_PGPASSWORD:-arcade_dev}" psql -h "$PGHOST" -p "$PGPORT" \ + -U "$PGUSER" -d "$scratch" -Atc \ + "SELECT COALESCE(SUM(balance_msat),0) FROM account_balances")" + else + total="$(docker compose exec -T "$COMPOSE_SERVICE" psql -U "$PGUSER" -d "$scratch" -Atc \ + "SELECT COALESCE(SUM(balance_msat),0) FROM account_balances")" + fi + + total="$(echo "$total" | tr -d '[:space:]')" + if [ "$total" = "0" ]; then + log "VERIFIED: restored ledger balances to zero" + else + die "restored ledger does NOT balance (sum = $total) — this backup is not trustworthy" + fi +} + +case "${1:-}" in + init) cmd_init ;; + sync) cmd_sync ;; + verify) cmd_verify ;; + *) echo "usage: $0 {init|sync|verify}" >&2; exit 2 ;; +esac diff --git a/ops/standby.sh b/ops/standby.sh new file mode 100755 index 0000000..9f939af --- /dev/null +++ b/ops/standby.sh @@ -0,0 +1,187 @@ +#!/usr/bin/env bash +# +# Warm standby: a second machine that can take over when the live one dies. +# +# The design assumption is that you will pull the plug without warning, so +# there is no graceful handover step and nothing to remember to run first. The +# standby continuously restores the latest backup and waits. Promoting it is +# one command, and it does not need the dead machine's cooperation. +# +# What this protects: the ledger, accounts, rounds, and fee history. +# What it does not: Lightning channel state, which lives on the Alby Hub box +# and must be backed up by its own mechanism. Restoring a stale channel state +# can lose funds — see ops/README.md before touching it. +# +# ./ops/standby.sh follow keep restoring the newest backup (run as a service) +# ./ops/standby.sh status how far behind the standby is +# ./ops/standby.sh promote become live +# +set -euo pipefail + +BACKUP_ROOT="${ARCADE_BACKUP_DIR:-/var/backups/quantum-arcade}" +DUMP_DIR="$BACKUP_ROOT/dumps" +BASE_DIR="$BACKUP_ROOT/base" +STATE="$BACKUP_ROOT/standby-restored" +PGUSER="${ARCADE_PGUSER:-arcade}" +PGDATABASE="${ARCADE_PGDATABASE:-arcade}" +COMPOSE_SERVICE="${ARCADE_PG_SERVICE:-postgres}" +INTERVAL="${ARCADE_STANDBY_INTERVAL:-300}" + +log() { printf '%s %s\n' "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$*"; } +die() { log "ERROR: $*" >&2; exit 1; } + +psql_cmd() { + if command -v psql >/dev/null 2>&1; then + PGPASSWORD="${ARCADE_PGPASSWORD:-arcade_dev}" \ + psql -h "${ARCADE_PGHOST:-localhost}" -U "$PGUSER" "$@" + else + docker compose exec -T "$COMPOSE_SERVICE" psql -U "$PGUSER" "$@" + fi +} + +restore_cmd() { + local db="$1" file="$2" + if command -v pg_restore >/dev/null 2>&1; then + PGPASSWORD="${ARCADE_PGPASSWORD:-arcade_dev}" \ + pg_restore -h "${ARCADE_PGHOST:-localhost}" -U "$PGUSER" \ + -d "$db" --clean --if-exists "$file" 2>/dev/null || true + else + docker compose exec -T "$COMPOSE_SERVICE" \ + pg_restore -U "$PGUSER" -d "$db" --clean --if-exists < "$file" 2>/dev/null || true + fi +} + +newest_backup() { + find "$DUMP_DIR" "$BASE_DIR" -name '*.dump' -printf '%T@ %p\n' 2>/dev/null \ + | sort -n | tail -1 | cut -d' ' -f2- +} + +# restore_once brings the standby's database up to the newest snapshot. +# +# It restores into a shadow database and swaps names on success. Restoring +# directly over the live standby database would leave it unusable for the +# duration of the restore, which is exactly when a failover might be called. +restore_once() { + local src + src="$(newest_backup)" + [ -n "$src" ] || { log "no backup available yet"; return 0; } + + local marker + marker="$(stat -c %Y "$src")" + if [ -f "$STATE" ] && [ "$(cat "$STATE")" = "$marker" ]; then + return 0 # already holding this snapshot + fi + + log "restoring $(basename "$src")" + local shadow="${PGDATABASE}_shadow" + + psql_cmd -d postgres -c "DROP DATABASE IF EXISTS $shadow" >/dev/null 2>&1 || true + psql_cmd -d postgres -c "CREATE DATABASE $shadow" >/dev/null + restore_cmd "$shadow" "$src" + + # The restored ledger must balance before it is allowed to become the + # standby's live copy. Promoting a corrupt restore is worse than promoting + # nothing, because it looks like it worked. + local total + total="$(psql_cmd -d "$shadow" -Atc \ + "SELECT COALESCE(SUM(balance_msat),0) FROM account_balances" | tr -d '[:space:]')" + if [ "$total" != "0" ]; then + psql_cmd -d postgres -c "DROP DATABASE IF EXISTS $shadow" >/dev/null 2>&1 || true + die "restored ledger does not balance (sum = $total); standby left on its previous copy" + fi + + # Swap: the previous copy becomes the fallback, the new one becomes current. + psql_cmd -d postgres -c "DROP DATABASE IF EXISTS ${PGDATABASE}_previous" >/dev/null 2>&1 || true + psql_cmd -d postgres -c \ + "ALTER DATABASE $PGDATABASE RENAME TO ${PGDATABASE}_previous" >/dev/null 2>&1 || true + psql_cmd -d postgres -c "ALTER DATABASE $shadow RENAME TO $PGDATABASE" >/dev/null + + echo "$marker" > "$STATE" + log "standby now holds $(basename "$src"), ledger verified" +} + +cmd_follow() { + log "following $DUMP_DIR every ${INTERVAL}s" + while true; do + restore_once || log "restore failed; keeping previous copy and retrying" + sleep "$INTERVAL" + done +} + +cmd_status() { + local src + src="$(newest_backup)" + if [ -z "$src" ]; then + echo " no backups present" + return 1 + fi + local age_backup age_restore now + now="$(date -u +%s)" + age_backup=$(( now - $(stat -c %Y "$src") )) + + printf ' newest backup %s\n' "$(basename "$src")" + printf ' backup age %ds\n' "$age_backup" + + if [ -f "$STATE" ]; then + age_restore=$(( now - $(cat "$STATE") )) + printf ' standby restored %ds behind live\n' "$age_restore" + else + printf ' standby restored never\n' + fi + + # A standby further behind than two backup intervals is not a standby. + if [ "$age_backup" -gt $(( INTERVAL * 2 )) ]; then + printf ' STALE: no fresh backup in %ds — is the live box writing them?\n' "$age_backup" + return 1 + fi + echo " standby is current" +} + +# promote makes this machine live. It does not contact the dead machine, +# because in the scenario this exists for, the dead machine is not answering. +cmd_promote() { + log "promoting this machine to live" + + restore_once || log "could not fetch a newer backup; promoting what is held" + + local total + total="$(psql_cmd -d "$PGDATABASE" -Atc \ + "SELECT COALESCE(SUM(balance_msat),0) FROM account_balances" | tr -d '[:space:]')" + [ "$total" = "0" ] || die "ledger does not balance (sum = $total); refusing to promote" + + local players rounds + players="$(psql_cmd -d "$PGDATABASE" -Atc \ + "SELECT count(*) FROM accounts WHERE kind='player'" | tr -d '[:space:]')" + rounds="$(psql_cmd -d "$PGDATABASE" -Atc \ + "SELECT COALESCE(max(id),0) FROM rounds" | tr -d '[:space:]')" + + log "ledger verified: $players players, latest round $rounds" + + docker compose up -d >/dev/null + log "arcade started" + + cat <&2; exit 2 ;; +esac diff --git a/ops/tune-kernel.sh b/ops/tune-kernel.sh new file mode 100755 index 0000000..ff2bd21 --- /dev/null +++ b/ops/tune-kernel.sh @@ -0,0 +1,180 @@ +#!/usr/bin/env bash +# +# Kernel and limit tuning for a machine running the arcade. +# +# Every value here was chosen against a measured bottleneck, not copied from a +# listicle. The measurements are in docs/SCALING.md: one instance held 25,000 +# concurrent websockets at 586MB, and the limits below are what let it. +# +# sudo ./ops/tune-kernel.sh apply write settings and reload +# ./ops/tune-kernel.sh show print current values +# ./ops/tune-kernel.sh check report values that are below target +# +set -euo pipefail + +CONF=/etc/sysctl.d/99-quantum-arcade.conf +LIMITS=/etc/security/limits.d/99-quantum-arcade.conf + +log() { printf '%s %s\n' "$(date -u +%H:%M:%SZ)" "$*"; } + +write_sysctl() { + cat > "$CONF" <<'EOF' +# Quantum Arcade — kernel tuning. +# +# Each setting exists because a specific limit was hit. Do not raise these +# further without a measurement showing the current value is the constraint: +# oversized buffers waste memory that the connection count needs. + +# --- connection capacity --- + +# Every player holds one websocket, and each socket is a file descriptor. +# 25k connections plus database pool, logs, and headroom. +fs.file-max = 2097152 +fs.nr_open = 2097152 + +# The listen backlog. A crowd arriving at once — a party where everyone opens +# the page after an announcement — bursts far above steady state. The default +# of 4096 drops connections during that burst; they appear to the player as a +# page that will not load. +net.core.somaxconn = 65535 +net.ipv4.tcp_max_syn_backlog = 65535 +net.core.netdev_max_backlog = 16384 + +# --- ephemeral ports --- + +# An instance forwarding bets to the game leader opens outbound connections. +# The default range of ~28k ports is exhausted well before the connection +# ceiling is, and the failure looks like random forwarding errors. +net.ipv4.ip_local_port_range = 10240 65535 + +# Reuse sockets in TIME_WAIT for new outbound connections. Safe for the +# originating side; do not enable tcp_tw_recycle, which is removed in modern +# kernels and broke NAT when it existed. +net.ipv4.tcp_tw_reuse = 1 +net.ipv4.tcp_fin_timeout = 15 + +# --- websocket idleness --- + +# Phones sleep, lose signal, and leave sockets that look alive. Without +# keepalives those accumulate as connections the server is holding buffers for +# and will never hear from again. Detect within ~5 minutes rather than 2 hours. +net.ipv4.tcp_keepalive_time = 240 +net.ipv4.tcp_keepalive_intvl = 30 +net.ipv4.tcp_keepalive_probes = 6 + +# --- buffers --- + +# Frames are ~1.8KB and sent a few times a second, so per-socket buffers can +# stay modest. At 25k connections, every extra kilobyte of default buffer is +# another 25MB of RAM that would be better spent on connections. +net.ipv4.tcp_rmem = 4096 87380 6291456 +net.ipv4.tcp_wmem = 4096 65536 6291456 +net.core.rmem_max = 12582912 +net.core.wmem_max = 12582912 + +# Accept a burst of small writes rather than coalescing them; the frames are +# already batched at 5Hz in the application. +net.ipv4.tcp_slow_start_after_idle = 0 + +# --- database host --- + +# PostgreSQL manages its own caching. Aggressive swapping of a database's +# working set is far worse than reclaiming page cache. +vm.swappiness = 10 +vm.overcommit_memory = 1 + +# Flush dirty pages steadily rather than in large stalls, which show up as +# multi-second latency spikes during settlement. +vm.dirty_background_ratio = 5 +vm.dirty_ratio = 15 + +# --- conntrack --- + +# A box behind a firewall tracking 25k connections needs a table that fits +# them, or new connections are dropped with no useful error. +net.netfilter.nf_conntrack_max = 262144 +EOF + log "wrote $CONF" +} + +write_limits() { + cat > "$LIMITS" <<'EOF' +# Quantum Arcade — process limits. +# +# fs.file-max raises the system ceiling; this raises the per-process one. +# Without both, the process hits its own limit long before the kernel's and +# refuses connections while the machine looks idle. +* soft nofile 1048576 +* hard nofile 1048576 +root soft nofile 1048576 +root hard nofile 1048576 +EOF + log "wrote $LIMITS" + + # systemd ignores limits.conf for services it starts. + mkdir -p /etc/systemd/system.conf.d + cat > /etc/systemd/system.conf.d/99-quantum-arcade.conf <<'EOF' +[Manager] +DefaultLimitNOFILE=1048576 +EOF + log "wrote systemd DefaultLimitNOFILE" +} + +cmd_apply() { + [ "$(id -u)" -eq 0 ] || { echo "must run as root" >&2; exit 1; } + write_sysctl + write_limits + sysctl --system >/dev/null + log "settings applied; reboot or re-login for limits to take effect" + cmd_check +} + +cmd_show() { + for k in fs.file-max net.core.somaxconn net.ipv4.ip_local_port_range \ + net.ipv4.tcp_keepalive_time vm.swappiness; do + printf ' %-34s %s\n' "$k" "$(sysctl -n "$k" 2>/dev/null || echo 'n/a')" + done + printf ' %-34s %s\n' "ulimit -n (this shell)" "$(ulimit -n)" +} + +# check reports what is below target, so a machine can be inspected without +# changing anything. +cmd_check() { + local fails=0 + check_min() { + local key="$1" want="$2" cur + cur="$(sysctl -n "$key" 2>/dev/null | awk '{print $1}')" || cur=0 + if [ -z "$cur" ] || [ "$cur" -lt "$want" ] 2>/dev/null; then + printf ' BELOW TARGET %-30s %s (want >= %s)\n' "$key" "${cur:-unset}" "$want" + fails=$((fails + 1)) + else + printf ' ok %-30s %s\n' "$key" "$cur" + fi + } + + check_min fs.file-max 1048576 + check_min net.core.somaxconn 32768 + check_min net.ipv4.tcp_max_syn_backlog 32768 + + local nofile + nofile="$(ulimit -n)" + if [ "$nofile" -lt 65536 ]; then + printf ' BELOW TARGET %-30s %s (want >= 65536)\n' "ulimit -n" "$nofile" + fails=$((fails + 1)) + else + printf ' ok %-30s %s\n' "ulimit -n" "$nofile" + fi + + if [ "$fails" -gt 0 ]; then + log "$fails setting(s) below target — run 'sudo $0 apply'" + return 1 + fi + log "all checked settings meet target" +} + +case "${1:-show}" in + apply) cmd_apply ;; + show) cmd_show ;; + check) cmd_check ;; + *) echo "usage: $0 {apply|show|check}" >&2; exit 2 ;; +esac