diff --git a/packages/footprint/src/cloud.js b/packages/footprint/src/cloud.js index fef359da..bcb324ef 100644 --- a/packages/footprint/src/cloud.js +++ b/packages/footprint/src/cloud.js @@ -50,6 +50,15 @@ export const CLOUD_SOURCES = { .map((p) => p.ipv4Prefix); }, }, + oracle: { + url: 'https://docs.oracle.com/iaas/tools/public_ip_ranges.json', + parse(json, region) { + return (json?.regions ?? []) + .filter((r) => !region || String(r.region ?? '').includes(region)) + .flatMap((r) => (r.cidrs ?? []).map((c) => c.cidr)) + .filter(Boolean); + }, + }, digitalocean: { url: 'https://www.digitalocean.com/geo/google.csv', csv: true, @@ -69,9 +78,27 @@ export const CLOUD_SOURCES = { export const SINGAPORE_REGIONS = { aws: 'ap-southeast-1', gcp: 'asia-southeast1', + oracle: 'ap-singapore', digitalocean: 'SG', }; +/** + * NOT COVERED, and worth knowing before trusting a miss. + * + * Azure publishes its service tags as a weekly file whose URL carries the + * date, so there is no stable address to fetch — it needs the download page + * scraped or the ARM API called with credentials. + * + * Alibaba Cloud publishes nothing at all. Reaching it means resolving its + * ASNs (AS45102 and friends) to prefixes through a BGP data source, which is + * a different kind of dependency from "the provider says these are ours". + * + * Both matter for Singapore specifically, where Alibaba is a common host for + * exactly this traffic. A caller that finds nothing here has NOT established + * that an address is residential. + */ +export const UNCOVERED_PROVIDERS = ['azure', 'alibaba', 'huawei', 'tencent']; + /** * Fetch published ranges. * diff --git a/packages/footprint/test/cloud.test.js b/packages/footprint/test/cloud.test.js index 32fbd471..25b4da33 100644 --- a/packages/footprint/test/cloud.test.js +++ b/packages/footprint/test/cloud.test.js @@ -199,5 +199,22 @@ describe('createCloudMatcher', () => { it('publishes the real provider URLs', () => { expect(CLOUD_SOURCES.aws.url).toContain('ip-ranges.amazonaws.com'); expect(CLOUD_SOURCES.gcp.url).toContain('gstatic.com/ipranges'); + expect(CLOUD_SOURCES.oracle.url).toContain('oracle.com'); + }); + + it('parses Oracle regions', async () => { + const body = { + regions: [ + { region: 'ap-singapore-1', cidrs: [{ cidr: '140.238.0.0/16' }] }, + { region: 'us-ashburn-1', cidrs: [{ cidr: '129.213.0.0/16' }] }, + ], + }; + const fetchImpl = vi.fn(async () => ({ ok: true, status: 200, json: async () => body })); + const { cidrs } = await fetchCloudRanges({ + providers: ['oracle'], + regions: { oracle: 'ap-singapore' }, + fetch: fetchImpl, + }); + expect(cidrs).toEqual(['140.238.0.0/16']); }); }); diff --git a/src/lib/cloud-gate.test.ts b/src/lib/cloud-gate.test.ts index 94def82d..a825a8fd 100644 --- a/src/lib/cloud-gate.test.ts +++ b/src/lib/cloud-gate.test.ts @@ -90,6 +90,25 @@ describe('judgeCloudClient', () => { } }); + it('never charges an uptime monitor', async () => { + // Not politeness. A monitor counts 200-399 as up, so a 402 does not bill + // it — it reports the site as DOWN. Ours runs on Railway, from a cloud + // address, so without this we page ourselves. + for (const ua of ['CrawlProof-Uptime/1.0', 'UptimeRobot/2.0', 'Pingdom.com_bot', 'Better Uptime Bot']) { + expect( + judgeCloudClient(req({ 'x-forwarded-for': CLOUD_IP, 'user-agent': ua }), '/'), + ).toMatchObject({ charge: false, reason: 'uptime monitor' }); + } + + const answer = await cloudGate( + new Request('https://coinpayportal.com/', { + headers: { 'x-forwarded-for': CLOUD_IP, 'user-agent': 'CrawlProof-Uptime/1.0' }, + }), + '/', + ); + expect(answer).toBeNull(); + }); + it('never charges a signed-in customer', () => { const v = judgeCloudClient( req({ 'x-forwarded-for': CLOUD_IP, cookie: 'sb-abcdef-auth-token=xyz' }), diff --git a/src/lib/cloud-gate.ts b/src/lib/cloud-gate.ts index c2f7bf17..d980c06a 100644 --- a/src/lib/cloud-gate.ts +++ b/src/lib/cloud-gate.ts @@ -83,7 +83,7 @@ const REFRESH_MS = 6 * 60 * 60 * 1000; * about one operator rather than about the behaviour. Anyone running the same * thing from Frankfurt is doing the same thing. */ -const matcher = createCloudMatcher({ providers: ['aws', 'gcp', 'digitalocean'] }); +const matcher = createCloudMatcher({ providers: ['aws', 'gcp', 'oracle', 'digitalocean'] }); let refreshing: Promise | null = null; let nextRefresh = 0; @@ -118,6 +118,20 @@ function ensureFresh(): void { /** Crawlers that send readers back. Never charged. */ const SEARCH_CRAWLER = /googlebot|bingbot|duckduckbot|applebot|yandex|baiduspider|slurp|oai-searchbot|chatgpt-user|perplexitybot|claude-searchbot|claude-user/i; +/** + * Uptime checkers, which must never be charged. + * + * Not politeness — correctness. A monitor counts 200-399 as up (CrawlProof's + * own `checkHttp` does exactly that), so answering it 402 does not bill + * anybody, it reports THE SITE AS DOWN. We would have built ourselves a false + * alarm generator and then been paged by it. + * + * Ours runs on Railway, which is to say from a cloud address, so the range + * check alone would catch it. Third-party monitors are listed for the same + * reason: whoever is watching this site is not the traffic we are pricing. + */ +const UPTIME_MONITOR = /crawlproof[\s_-]?uptime|uptime[\s_-]?robot|pingdom|statuscake|better[\s_-]?uptime|hetrixtool|site24x7|newrelic|datadog|checkly|updown\.io|freshping|cron-job\.org|monitoring|healthcheck/i; + function isSignedIn(request: Request): boolean { const cookie = request.headers.get('cookie') ?? ''; return /sb-[^=]*auth-token=/.test(cookie) || /coinpay_session=/.test(cookie); @@ -150,6 +164,7 @@ export function judgeCloudClient(request: Request, pathname: string): CloudVerdi const ua = request.headers.get('user-agent') ?? ''; if (SEARCH_CRAWLER.test(ua)) return { charge: false, reason: 'search crawler' }; + if (UPTIME_MONITOR.test(ua)) return { charge: false, reason: 'uptime monitor' }; if (isSignedIn(request)) return { charge: false, reason: 'signed in' }; if (!matcher.matches(ip)) return { charge: false, reason: 'not a published cloud range' };