diff --git a/.github/workflows/deploy-prober.yml b/.github/workflows/deploy-prober.yml new file mode 100644 index 00000000..0cdee59c --- /dev/null +++ b/.github/workflows/deploy-prober.yml @@ -0,0 +1,65 @@ +name: deploy-prober + +# Deploys the port-drift prober (docs/uptime-monitoring-prd.md §12) to the +# DigitalOcean droplet ubuntu@scan.crawlproof.com on merges to master. +# +# FULLY SELF-BOOTSTRAPPING + IDEMPOTENT: the workflow installs every system +# dependency (nmap, Node 20, build tools), writes the Redis env file, installs +# the systemd unit, builds, and (re)starts the service on EVERY run. First run +# provisions a bare droplet; later runs just update. You never SSH in to set up. +# +# The prober connects OUTBOUND to Redis — nothing here opens an inbound port on +# the droplet (only sshd, which already exists). +# +# Required repo secrets: +# DROPLET_SSH_KEY - private SSH key for ubuntu@scan.crawlproof.com +# PROBER_REDIS_URL - rediss://... broker URL (scoped to the "prober" queue) + +on: + push: + branches: [master] + paths: + - "prober/**" + - "lib/**" # prober shares job-payload types from lib/ + - ".github/workflows/deploy-prober.yml" + workflow_dispatch: # allow manual redeploys + +# Never let two deploys race on the droplet. +concurrency: + group: deploy-prober + cancel-in-progress: false + +env: + DEPLOY_HOST: scan.crawlproof.com + DEPLOY_USER: ubuntu + DEPLOY_PATH: /home/ubuntu/crawlproof.com + +jobs: + deploy: + name: provision + deploy prober + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - name: Configure SSH + run: | + install -m 700 -d ~/.ssh + printf '%s\n' "${{ secrets.DROPLET_SSH_KEY }}" > ~/.ssh/id_deploy + chmod 600 ~/.ssh/id_deploy + ssh-keyscan -H "$DEPLOY_HOST" >> ~/.ssh/known_hosts 2>/dev/null + + - name: Ship prober + lib to droplet + run: | + ssh -i ~/.ssh/id_deploy "$DEPLOY_USER@$DEPLOY_HOST" "mkdir -p '$DEPLOY_PATH'" + # --delete keeps the droplet in sync; node_modules/dist are excluded so + # they are neither shipped nor deleted (rebuilt on the box). + rsync -az --delete \ + -e "ssh -i ~/.ssh/id_deploy" \ + --exclude node_modules --exclude dist \ + prober lib \ + "$DEPLOY_USER@$DEPLOY_HOST:$DEPLOY_PATH/" + + - name: Provision + restart (idempotent) + run: | + ssh -i ~/.ssh/id_deploy "$DEPLOY_USER@$DEPLOY_HOST" \ + "REDIS_URL='${{ secrets.PROBER_REDIS_URL }}' DEPLOY_PATH='$DEPLOY_PATH' bash '$DEPLOY_PATH/prober/deploy/provision.sh'" diff --git a/docs/uptime-monitoring-prd.md b/docs/uptime-monitoring-prd.md index 4801c9e1..98615665 100644 --- a/docs/uptime-monitoring-prd.md +++ b/docs/uptime-monitoring-prd.md @@ -57,7 +57,7 @@ On-call escalation is explicitly a later phase. ## 2. Objectives ### 2.1 Primary Goals -- Detect downtime for HTTP(S), TCP, PING, keyword, and SSL-expiry checks. +- Detect downtime for HTTP(S), TCP, keyword, and SSL-expiry checks (ICMP PING is Phase 2 — see §12). - Alert within one confirmed check cycle across multiple channels; alert on recovery. - Free tier of **20 monitors** at 60s interval, no credit card. - Public status page per project (uptime %, response times, incidents). @@ -84,8 +84,8 @@ On-call escalation is explicitly a later phase. | **HTTP(S)** | Status code, response time, redirect handling | URL, expected status, timeout, follow-redirects | | **Keyword** | HTTP body contains / omits a string | URL, keyword, match mode | | **SSL expiry** | Cert days-to-expiry warning | Host, warn-days (default 14) | -| **TCP** | Port open | Host, port | -| **PING (ICMP)** | Host reachability | Host / IP | +| **TCP** | Port open (outbound `connect()`, works from Railway) | Host, port | +| **PING (ICMP)** *(Phase 2)* | Host reachability — needs raw sockets, so runs on the §12 prober droplet, not Railway | Host / IP | Each monitor: name, type, target, interval (60s free / 30s paid), timeout, expected-result config, channel(s), enabled flag, and optional link to the @@ -212,8 +212,8 @@ All RLS-scoped to org/project, consistent with existing tables. | **M1 — Core loop** | HTTP + keyword + SSL monitors, due-time sweep in worker, state machine, email + webhook alerts, monitor CRUD UI | | **M2 — Channels + status page** | Slack, Discord, public status page, incident history, uptime % | | **M3 — Plans + limits** | Fold into existing plan tiers, 20-free enforcement, SMS (Twilio) + caps, custom domains | -| **M4 — Polish** | TCP/PING checks, maintenance windows, schedule gates, weekly summary email | -| **Phase 2** | On-call escalation policies, multi-region probing, PagerDuty/Teams/Jira, Playwright journeys | +| **M4 — Polish** | TCP checks, maintenance windows, schedule gates, weekly summary email | +| **Phase 2** | On-call escalation policies, multi-region probing, PagerDuty/Teams/Jira, Playwright journeys, **ICMP PING checks (via §12 prober droplet)**, **exposed-services / port-drift check (§12)** | --- @@ -224,5 +224,92 @@ All RLS-scoped to org/project, consistent with existing tables. optional `project_id`, counts against org limit.) 3. Fold into existing CrawlProof plans, or introduce an uptime add-on SKU? 4. Status page: reuse existing public-report subdomain scheme, or new namespace? -5. Do TCP/PING (ICMP) checks work from the Railway runtime, or do they need an - external prober? (May push PING to Phase 2.) +5. **Resolved:** TCP works from Railway (outbound `connect()`); **ICMP PING does + not** (needs raw sockets), so PING moves to **Phase 2** and runs on the §12 + prober droplet alongside the port-drift scans. nmap is also Phase 2 (§12). + +--- + +## 12. Phase 2 — Exposed-Services / Port-Drift Check + +A lightweight attack-surface monitor: on a verified-owned host, detect ports that +are open to the public internet but *shouldn't* be — e.g. a Redis (6379) or +Postgres (5432) accidentally exposed. Framed as **security drift detection**, not +a scanner. + +> **Value framing:** "You told us to watch 80/443. We now also see **6379 (Redis)** +> reachable from the public internet on your verified host — likely a +> misconfiguration." Alerts fire when the open-port set *changes* from an accepted +> baseline, not merely because a port is open. + +### 12.1 Hard constraints (why this is Phase 2, not V1) +- **Owned targets only.** Reuse CrawlProof's existing **domain-ownership + verification** as the gate. No arbitrary hostnames — this must never become + port-scanning-as-a-service against third parties. +- **Not from the Railway app IP.** Scans run on a **self-hosted prober (a + DigitalOcean droplet)**, never from the Railway service, so an egress-abuse flag + can't take down production. GCP and Railway AUPs prohibit network scanning; + scanning from the app IP risks the service's IP. The droplet has its own IP + reputation and raw-socket access. See §12.3 for the job-transport design. +- **Bounded + TCP-connect only.** Curated **top-~100 common service ports**, never + full 65535; TCP `connect()` only (containers lack `CAP_NET_RAW` for SYN scans + anyway); ICMP discovery skipped (`-Pn`-equivalent). +- **Rate-limited + infrequent.** Daily cadence, not per-minute; per-org caps. + +### 12.2 Behavior +- Establish a **baseline** open-port set per host on first scan (user confirms + "these are expected"). +- On each subsequent scan, diff against baseline. **New open port → alert** + (down-alert-style, same channels). Closed expected port → optional info alert. +- Record findings as incidents/events so they show in history and (optionally) a + private security view — **not** the public status page by default. + +### 12.3 Prober architecture — BullMQ over Redis, prober stays in the monorepo + +Job transport is a **BullMQ queue backed by Redis**, not synchronous RPC (scans +take seconds-to-minutes and must survive restarts). BullMQ gives us retries + +backoff, concurrency control, and **repeatable jobs** for the daily cadence — so +no hand-rolled leasing/reaper is needed. + +``` +Railway (producer) Redis (broker) DO droplet (prober) +────────────────── ────────────── ─────────────────── +API / worker enqueues ──▶ "prober" queue ◀── BullMQ Worker dials OUT +port-scan jobs (rediss:// TLS) over rediss://, runs +(repeatable = daily) nmap -sT -Pn --top-ports + 100, returns result +Railway QueueEvents ◀─── job "completed" ◀── (result = job return value) +handler persists to +Supabase, diffs +baseline, fires alerts +``` + +- **Redis is new infra.** CrawlProof has no Redis today (the worker uses + in-container loopback HTTP). Add one (Railway Redis plugin or Upstash) exposed + over `rediss://` with auth. The droplet connects **outbound only** to Redis — + no inbound port on the prober box (firewall to egress 443/6380 only). +- **Prober lives in the monorepo** as a new workspace (mirrors the existing + `worker/`), e.g. `prober/` — its own `package.json` and entrypoint, sharing the + **job-payload types from `lib/`** so producer and prober can't drift. +- **Deploy is self-bootstrapping.** The `deploy-prober` GitHub Action fires on + merges to `master` (droplet `ubuntu@scan.crawlproof.com`): rsync `prober/`+`lib/` + to the box, then run `prober/deploy/provision.sh` — an **idempotent** script + that installs nmap/Node/build tools, writes the Redis env file, installs the + `crawlproof-prober` systemd unit, builds, and restarts. First run provisions a + bare droplet; later runs update. **No manual SSH/setup on the droplet, ever.** +- **Results flow back via the job return value** — the droplet writes no DB. A + Railway-side `QueueEvents` (or a second in-process Worker) handles `completed` + /`failed`, then persists results, diffs the baseline, and dispatches alerts. + This keeps Supabase credentials off the droplet. +- **Credential blast radius:** the droplet holds only the `rediss://` URL. Scope + it with a **dedicated Redis instance (or ACL user)** limited to the `prober` + queue keys, so a compromised droplet can't read other queues. +- **Repeatable jobs** drive the daily per-monitor scan schedule; per-org caps and + `attempts`/backoff are BullMQ config, not custom code. + +### 12.4 Open items +- Which port list ships as the default "watch" set. +- Whether findings feed CrawlProof's existing audit/finding surface or a new one. +- Redis host choice (Railway plugin vs. Upstash) and TLS/ACL setup. +- One droplet = single vantage point + SPOF (fine for daily scans); the BullMQ + Worker model scales to N droplets with zero config change if redundancy is wanted. diff --git a/prober/deploy/crawlproof-prober.service b/prober/deploy/crawlproof-prober.service new file mode 100644 index 00000000..f9830efb --- /dev/null +++ b/prober/deploy/crawlproof-prober.service @@ -0,0 +1,27 @@ +# systemd unit for the CrawlProof port-drift prober (docs/uptime-monitoring-prd.md §12). +# Installed and (re)started automatically by prober/deploy/provision.sh — you do +# NOT install this by hand. Layout matches the deploy-prober GitHub Action: +# repo checkout : /home/ubuntu/crawlproof.com +# Redis env : /home/ubuntu/crawlproof-prober.env (0600, holds REDIS_URL) + +[Unit] +Description=CrawlProof port-drift prober (BullMQ worker) +After=network-online.target +Wants=network-online.target + +[Service] +Type=simple +User=ubuntu +WorkingDirectory=/home/ubuntu/crawlproof.com/prober +EnvironmentFile=/home/ubuntu/crawlproof-prober.env +ExecStart=/usr/bin/node dist/index.js +Restart=always +RestartSec=5 +# Hardening — the prober needs only outbound network + nmap. +NoNewPrivileges=true +ProtectSystem=strict +PrivateTmp=true +ReadWritePaths=/home/ubuntu/crawlproof.com/prober + +[Install] +WantedBy=multi-user.target diff --git a/prober/deploy/provision.sh b/prober/deploy/provision.sh new file mode 100644 index 00000000..edee6bd7 --- /dev/null +++ b/prober/deploy/provision.sh @@ -0,0 +1,50 @@ +#!/usr/bin/env bash +# Idempotent provisioning for the CrawlProof port-drift prober droplet +# (docs/uptime-monitoring-prd.md §12). Run remotely by the deploy-prober GitHub +# Action on every deploy. Safe to re-run: performs first-time bootstrap of a +# bare droplet AND routine updates. No interactive login is ever required. +# +# Inputs (env): +# REDIS_URL - rediss://... broker URL (required) +# DEPLOY_PATH - repo checkout on the droplet (default /home/ubuntu/crawlproof.com) +set -euo pipefail + +DEPLOY_PATH="${DEPLOY_PATH:-/home/ubuntu/crawlproof.com}" +ENV_FILE="/home/ubuntu/crawlproof-prober.env" +UNIT_SRC="${DEPLOY_PATH}/prober/deploy/crawlproof-prober.service" +UNIT_DST="/etc/systemd/system/crawlproof-prober.service" +export DEBIAN_FRONTEND=noninteractive + +if [ -z "${REDIS_URL:-}" ]; then + echo "[provision] ERROR: REDIS_URL is not set" >&2 + exit 1 +fi + +echo "[provision] system packages (nmap, build tools)" +sudo apt-get update -y +sudo apt-get install -y --no-install-recommends nmap curl ca-certificates build-essential + +echo "[provision] Node 20" +if ! command -v node >/dev/null 2>&1 \ + || [ "$(node -p 'process.versions.node.split(".")[0]' 2>/dev/null || echo 0)" != "20" ]; then + curl -fsSL https://deb.nodesource.com/setup_20.x | sudo -E bash - + sudo apt-get install -y nodejs +fi + +echo "[provision] Redis env file (0600)" +umask 077 +printf 'REDIS_URL=%s\n' "${REDIS_URL}" > "${ENV_FILE}" +chmod 600 "${ENV_FILE}" + +echo "[provision] build prober" +cd "${DEPLOY_PATH}/prober" +npm ci --no-audit --no-fund +npm run build + +echo "[provision] install + (re)start systemd service" +sudo cp "${UNIT_SRC}" "${UNIT_DST}" +sudo systemctl daemon-reload +sudo systemctl enable crawlproof-prober +sudo systemctl restart crawlproof-prober +sudo systemctl --no-pager --lines=5 status crawlproof-prober || true +echo "[provision] done"