From 1348ebf30b88678bcd8edb49ec4d374f133f656d Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Sat, 5 Sep 2026 21:30:43 +0000 Subject: [PATCH] Charge AI training crawlers for access (@profullstack/x402-gateway) Training crawlers (GPTBot, ClaudeBot, CCBot, meta-externalagent, Bytespider, Applebot-Extended) get 402 Payment Required with an x402 offer, or the sales page at /crawl, and a paid pass opens the site for a day. People, search engines and retrieval crawlers pass through untouched. robots.txt is now generated from the same lists. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01YafYxayh7Gqe5MWNNQMev2 --- .env.example | 6 ++++++ app/robots.ts | 19 ------------------- app/robots.txt/route.ts | 9 +++++++++ lib/crawl-gateway.ts | 27 +++++++++++++++++++++++++++ package-lock.json | 10 ++++++++++ package.json | 1 + proxy.ts | 7 +++++++ 7 files changed, 60 insertions(+), 19 deletions(-) delete mode 100644 app/robots.ts create mode 100644 app/robots.txt/route.ts create mode 100644 lib/crawl-gateway.ts diff --git a/.env.example b/.env.example index 1cf198b1..f97bf824 100644 --- a/.env.example +++ b/.env.example @@ -114,3 +114,9 @@ GITHUB_APP_SLUG= # Playwright owner-flow e2e. Email defaults to anthony+riotcoder@profullstack.com. PLAYWRIGHT_AUTH_EMAIL= PLAYWRIGHT_AUTH_PASSWORD= + +# Crawl gateway (@profullstack/x402-gateway): AI training crawlers pay $1/day over +# x402. A SCOPED CoinPay key (payments:create) and the EVM address that receives +# the USDC. Unset = crawlers still get 402, nothing sold. +COINPAY_X402_KEY= +CRAWL_PAY_TO= diff --git a/app/robots.ts b/app/robots.ts deleted file mode 100644 index dc85cd08..00000000 --- a/app/robots.ts +++ /dev/null @@ -1,19 +0,0 @@ -import type { MetadataRoute } from "next"; -import { env } from "@/lib/env"; - -export default function robots(): MetadataRoute.Robots { - return { - rules: [ - { userAgent: "*", allow: "/" }, - { userAgent: "GPTBot", allow: "/" }, - { userAgent: "ClaudeBot", allow: "/" }, - { userAgent: "PerplexityBot", allow: "/" }, - { userAgent: "Google-Extended", allow: "/" }, - { userAgent: "OAI-SearchBot", allow: "/" }, - { userAgent: "Applebot-Extended", allow: "/" }, - { userAgent: "CCBot", allow: "/" }, - ], - sitemap: `${env.siteUrl}/sitemap.xml`, - host: env.siteUrl, - }; -} diff --git a/app/robots.txt/route.ts b/app/robots.txt/route.ts new file mode 100644 index 00000000..319d762f --- /dev/null +++ b/app/robots.txt/route.ts @@ -0,0 +1,9 @@ +import { robotsRoute } from "@profullstack/x402-gateway/next"; +import { gateway } from "@/lib/crawl-gateway"; + +// Generated from the same crawler lists the gateway enforces: training +// crawlers are refused everywhere but /crawl (where they can buy a pass), +// retrieval crawlers are named as welcome, everyone else gets the rules below. +export const GET = robotsRoute(gateway, { + disallow: ["/api/"], +}); diff --git a/lib/crawl-gateway.ts b/lib/crawl-gateway.ts new file mode 100644 index 00000000..6d0ad628 --- /dev/null +++ b/lib/crawl-gateway.ts @@ -0,0 +1,27 @@ +import { createGateway } from "@profullstack/x402-gateway"; +import { x402Proxy } from "@profullstack/x402-gateway/next"; + +/** + * Sells crawl access to AI training crawlers (GPTBot, ClaudeBot, CCBot, + * meta-externalagent, Bytespider, Applebot-Extended, ...) by the day over + * x402, settled by CoinPay in USDC. People, Googlebot and the retrieval + * crawlers behind AI search pass through untouched. + * + * Runs inside the middleware, so nothing here may import Node-only modules. + * The env is read through a non-literal key on purpose: Next inlines + * `process.env.NAME` at build time, and these are runtime secrets. Without + * COINPAY_X402_KEY and CRAWL_PAY_TO the gateway still answers training + * crawlers with 402, just with an empty offer. + */ +const env = (name: string) => process.env[name]; + +export const gateway = createGateway({ + siteUrl: env("SITE_URL") || env("NEXT_PUBLIC_SITE_URL") || "https://crawlproof.com", + siteName: "CrawlProof", + coinpay: { apiKey: env("COINPAY_X402_KEY") }, + payTo: env("CRAWL_PAY_TO"), + contact: "mailto:support@crawlproof.com", +}); + +/** Resolves to a Response for a refused crawler, or undefined to carry on. */ +export const gate = x402Proxy(gateway); diff --git a/package-lock.json b/package-lock.json index 8491f035..0fe2ab90 100644 --- a/package-lock.json +++ b/package-lock.json @@ -17,6 +17,7 @@ "@profullstack/referrals": "^0.1.0", "@profullstack/stack": "^0.1.3", "@profullstack/x402-client": "^0.2.0", + "@profullstack/x402-gateway": "^0.1.0", "@supabase/ssr": "^0.10.3", "@supabase/supabase-js": "^2.105.4", "@types/nodemailer": "^8.0.0", @@ -1730,6 +1731,15 @@ "node": ">=20.19" } }, + "node_modules/@profullstack/x402-gateway": { + "version": "0.1.0", + "resolved": "https://registry.npmjs.org/@profullstack/x402-gateway/-/x402-gateway-0.1.0.tgz", + "integrity": "sha512-B7tWvWk/bIEoqyec6UoyRF1pO7X/+b+wFRv2ZFIClqskmEpyxoA559ZgdTvnxqAIvuDeE9v56nVpYRQ+lmOZQQ==", + "license": "MIT", + "engines": { + "node": ">=20.11" + } + }, "node_modules/@react-email/render": { "version": "1.1.2", "resolved": "https://registry.npmjs.org/@react-email/render/-/render-1.1.2.tgz", diff --git a/package.json b/package.json index d20c9418..e5777619 100644 --- a/package.json +++ b/package.json @@ -29,6 +29,7 @@ "@profullstack/referrals": "^0.1.0", "@profullstack/stack": "^0.1.3", "@profullstack/x402-client": "^0.2.0", + "@profullstack/x402-gateway": "^0.1.0", "@supabase/ssr": "^0.10.3", "@supabase/supabase-js": "^2.105.4", "@types/nodemailer": "^8.0.0", diff --git a/proxy.ts b/proxy.ts index 3a877066..ff657cea 100644 --- a/proxy.ts +++ b/proxy.ts @@ -1,3 +1,4 @@ +import { gate } from "@/lib/crawl-gateway"; import { NextResponse, type NextRequest } from "next/server"; import { createServerClient, type CookieOptions } from "@supabase/ssr"; import { trackReferralCode } from "@profullstack/stack/referrals"; @@ -5,6 +6,12 @@ import { trackReferralCode } from "@profullstack/stack/referrals"; type Cookie = { name: string; value: string; options?: CookieOptions }; export async function proxy(request: NextRequest) { + // Crawl gateway first: AI training crawlers get 402 Payment Required (or the + // sales page at /crawl) unless they present a paid pass. People, Googlebot + // and retrieval crawlers fall through to everything below. + const answer = await gate(request); + if (answer) return answer; + // 308 redirect www.crawlproof.com -> crawlproof.com (preserves method + body). const host = request.headers.get("host") ?? ""; if (host.toLowerCase().startsWith("www.")) {