diff --git a/lib/outreach/discover.ts b/lib/outreach/discover.ts index 86f30cf5..73a0baad 100644 --- a/lib/outreach/discover.ts +++ b/lib/outreach/discover.ts @@ -21,6 +21,7 @@ import { searchSerp, hasValueSerpKey } from "@/lib/alerts/valueserp"; import { isThirdPartyHost } from "@/lib/leadCampaign"; import { businessSearch } from "./freeSearch"; import { normalizeHost } from "./cold"; +import { looksLikeLoginWall } from "./loginWall"; export type DiscoveredProspect = { host: string; @@ -178,7 +179,10 @@ function looksBlocked(status: number): boolean { async function fetchHtml( url: string, -): Promise<{ ok: true; html: string } | { ok: false; error: string; status?: number }> { +): Promise< + | { ok: true; html: string; finalUrl: string } + | { ok: false; error: string; status?: number; loginRequired?: boolean } +> { try { const res = await fetch(url, { headers: { "user-agent": SEED_UA }, @@ -186,7 +190,15 @@ async function fetchHtml( redirect: "follow", }); if (!res.ok) return { ok: false, status: res.status, error: `HTTP ${res.status}` }; - return { ok: true, html: await res.text() }; + const html = await res.text(); + // The redirect onto a login page is visible here even though the login + // form itself may only exist after scripts run, so this catches the wall + // without paying for a render. + const wall = looksLikeLoginWall({ requestedUrl: url, finalUrl: res.url, html }); + if (wall.loginRequired) { + return { ok: false, loginRequired: true, error: `the site requires a login — ${wall.reason}` }; + } + return { ok: true, html, finalUrl: res.url }; } catch (err) { return { ok: false, error: err instanceof Error ? err.message : "unknown" }; } @@ -204,8 +216,12 @@ async function fetchHtml( async function loadSeedHtml( url: string, allowRender: boolean, -): Promise<{ html: string; rendered: boolean } | { error: string }> { +): Promise<{ html: string; rendered: boolean } | { error: string; loginRequired?: boolean }> { const direct = await fetchHtml(url); + // A login wall is settled: rendering it again only renders the login page. + if (!direct.ok && direct.loginRequired) { + return { error: direct.error, loginRequired: true }; + } if (direct.ok) { const hasCandidates = extractOutboundProspects({ html: direct.html, sourceUrl: url, limit: 1 }).length > 0; if (hasCandidates || !allowRender) return { html: direct.html, rendered: false }; @@ -216,6 +232,7 @@ async function loadSeedHtml( const { renderPage } = await import("./render"); const rendered = await renderPage(url); if (rendered.ok) return { html: rendered.html, rendered: true }; + if (rendered.loginRequired) return { error: rendered.error, loginRequired: true }; // Prefer the render error: when both fail it is the more specific of the // two, and it distinguishes a bot challenge from an ordinary failure. @@ -242,14 +259,24 @@ export async function discoverFromSeed(input: { * businesses. Keeps ordinary directories at one cheap page load. */ detailHopThreshold?: number; -}): Promise<{ prospects: DiscoveredProspect[]; error?: string; notes?: string[] }> { +}): Promise<{ + prospects: DiscoveredProspect[]; + error?: string; + notes?: string[]; + /** Set when the seed was withheld pending a login, so the UI can offer to store one. */ + loginRequired?: boolean; +}> { const limit = input.limit ?? 100; const allowRender = input.render !== false; const notes: string[] = []; const seed = await loadSeedHtml(input.seedUrl, allowRender); if ("error" in seed) { - return { prospects: [], error: `seed ${input.seedUrl} failed: ${seed.error}` }; + return { + prospects: [], + error: `seed ${input.seedUrl} failed: ${seed.error}`, + loginRequired: seed.loginRequired, + }; } if (seed.rendered) notes.push(`rendered ${input.seedUrl} in a browser`); @@ -371,10 +398,17 @@ export async function discoverProspects(input: { seedUrls?: string[]; limit?: number; source?: SearchSource; -}): Promise<{ prospects: DiscoveredProspect[]; serpCalls: number; errors: string[] }> { +}): Promise<{ + prospects: DiscoveredProspect[]; + serpCalls: number; + errors: string[]; + /** Seeds that returned a login wall — the UI offers to store credentials for these. */ + loginRequiredSeeds: string[]; +}> { const limit = input.limit ?? 50; const merged = new Map(); const errors: string[] = []; + const loginRequiredSeeds: string[] = []; let serpCalls = 0; const queries = (input.queries ?? []).slice(0, 5); @@ -396,8 +430,9 @@ export async function discoverProspects(input: { // most often paste in. const res = await discoverFromSeed({ seedUrl, limit, depth: 2 }); if (res.error) errors.push(res.error); + if (res.loginRequired) loginRequiredSeeds.push(seedUrl); for (const p of res.prospects) if (!merged.has(p.host)) merged.set(p.host, p); } - return { prospects: [...merged.values()].slice(0, limit), serpCalls, errors }; + return { prospects: [...merged.values()].slice(0, limit), serpCalls, errors, loginRequiredSeeds }; } diff --git a/lib/outreach/loginWall.ts b/lib/outreach/loginWall.ts new file mode 100644 index 00000000..075b29cb --- /dev/null +++ b/lib/outreach/loginWall.ts @@ -0,0 +1,107 @@ +// Tell a login wall apart from an empty directory. +// +// Both look identical to seed discovery: HTTP 200, a page, and no business +// links on it. Reporting "no businesses found" for a site that simply wanted +// a login sends the user off to debug their search terms when the actual +// problem is that they were never shown the page. +// +// Instagram is the motivating case and a good illustration of why status +// codes are useless here — asking for a search page returns 200, having +// quietly redirected to /accounts/login/?next=. +// +// Kept dependency-free so both the fetch path and the render path can use it: +// a redirect is visible to plain fetch, while a password field on a +// JS-rendered login screen only appears once scripts have run. + +/** Paths a site sends you to when it wants credentials. */ +const LOGIN_PATH_RE = + /(^|\/)(accounts\/login|login|log-in|signin|sign-in|sign_in|auth\/login|users\/sign_in|session\/new|checkpoint)(\/|$|\?)/i; + +/** + * Query parameters a login page uses to remember where you were going. + * Their presence alongside a login path is what distinguishes "you were + * bounced here" from "you asked for the login page". + */ +const RETURN_PARAM_RE = /[?&](next|return_to|returnurl|redirect(_uri|_to)?|continue|dest|destination)=/i; + +const PASSWORD_INPUT_RE = /]+type=["']password["']/i; + +/** Phrases that accompany a credential form rather than ordinary page copy. */ +const LOGIN_COPY_RE = + /(forgot (your )?password|log in to continue|sign in to continue|please log in|login required|create an account to)/i; + +export type LoginWallVerdict = { + loginRequired: boolean; + /** Why we think so — surfaced to the user so the call is auditable. */ + reason: string | null; +}; + +/** + * Decide whether `html` (fetched from `finalUrl`, having been asked for + * `requestedUrl`) is a login wall. + * + * The strongest signal is a redirect off the requested page onto a login + * path, because that is the site explicitly refusing the request. A password + * field is next: ordinary directory pages don't carry one. Login copy alone + * is not enough — plenty of real pages have a "log in" link in the header — + * so it only counts as corroboration for a page that also moved us. + */ +export function looksLikeLoginWall(input: { + requestedUrl: string; + finalUrl?: string; + html?: string; +}): LoginWallVerdict { + const requested = safeUrl(input.requestedUrl); + const final = safeUrl(input.finalUrl ?? input.requestedUrl); + + const movedUs = + Boolean(final && requested) && + (final!.pathname !== requested!.pathname || final!.host !== requested!.host); + + const finalIsLoginPath = final ? LOGIN_PATH_RE.test(final.pathname) : false; + const carriesReturn = final ? RETURN_PARAM_RE.test(final.search) : false; + + // Redirected onto a login path: unambiguous. + if (movedUs && finalIsLoginPath) { + return { + loginRequired: true, + reason: carriesReturn + ? `redirected to ${final!.pathname} carrying the page you asked for as a return parameter` + : `redirected to ${final!.pathname}`, + }; + } + + const html = input.html ?? ""; + const hasPasswordField = PASSWORD_INPUT_RE.test(html); + + // A password field on the page we were served, when we asked for something + // that was not a login page. + if (hasPasswordField && !(requested && LOGIN_PATH_RE.test(requested.pathname))) { + return { + loginRequired: true, + reason: movedUs + ? `served a password field after redirecting to ${final?.pathname ?? "another page"}` + : "the page contains a password field", + }; + } + + // Moved us somewhere that reads like a login screen without matching a + // known login path — covers hosts with unusual routes. + if (movedUs && LOGIN_COPY_RE.test(html)) { + return { + loginRequired: true, + reason: `redirected to ${final?.pathname ?? "another page"}, which reads like a sign-in screen`, + }; + } + + return { loginRequired: false, reason: null }; +} + +function safeUrl(value: string | undefined): URL | null { + if (!value) return null; + try { + return new URL(value); + } catch { + return null; + } +} diff --git a/lib/outreach/render.ts b/lib/outreach/render.ts index 9999991e..4ba6fc8a 100644 --- a/lib/outreach/render.ts +++ b/lib/outreach/render.ts @@ -25,6 +25,7 @@ import type { Browser, BrowserContext } from "playwright"; import { isPrivateAddress } from "./mailboxDiscovery"; +import { looksLikeLoginWall } from "./loginWall"; import dns from "node:dns/promises"; import net from "node:net"; @@ -98,7 +99,14 @@ async function hostIsPublic(hostname: string): Promise { export type RenderResult = | { ok: true; html: string; status: number; finalUrl: string } - | { ok: false; error: string; status?: number; challenged?: boolean }; + | { + ok: false; + error: string; + status?: number; + challenged?: boolean; + /** The page was withheld pending a login, not missing. */ + loginRequired?: boolean; + }; /** Page titles a bot-protection interstitial serves instead of the real page. */ const CHALLENGE_RE = /just a moment|attention required|verifying you are human|checking your browser/i; @@ -172,10 +180,25 @@ export async function renderPage(url: string): Promise { } const html = await page.content(); + const finalUrl = page.url(); + + // A login wall answers 200 with a real page, so it has to be caught by + // what the page is rather than by the status. Reported as a failure + // because the caller asked for a directory and did not get one. + const wall = looksLikeLoginWall({ requestedUrl: url, finalUrl, html }); + if (wall.loginRequired) { + return { + ok: false, + status, + loginRequired: true, + error: `the site requires a login — ${wall.reason}`, + }; + } + return { ok: true, status, - finalUrl: page.url(), + finalUrl, html: html.length > MAX_HTML_BYTES ? html.slice(0, MAX_HTML_BYTES) : html, }; } catch (error) { diff --git a/tests/login-wall.test.ts b/tests/login-wall.test.ts new file mode 100644 index 00000000..1eb4014d --- /dev/null +++ b/tests/login-wall.test.ts @@ -0,0 +1,93 @@ +import { describe, it, expect } from "vitest"; +import { looksLikeLoginWall } from "@/lib/outreach/loginWall"; + +// The exact redirect Instagram served for a seed search URL: HTTP 200, but +// bounced onto the login page with the requested URL preserved in ?next=. +const IG_REQUESTED = + "https://www.instagram.com/explore/search/keyword/?q=military%203d%20modeling%20games"; +const IG_FINAL = + "https://www.instagram.com/accounts/login/?next=https%3A%2F%2Fwww.instagram.com%2Fexplore%2Fsearch%2Fkeyword%2F%3Fq%3Dmilitary%2B3d%2Bmodeling%2Bgames%26__coig_login%3D1"; + +describe("looksLikeLoginWall", () => { + it("catches the Instagram redirect that reported as 'no businesses found'", () => { + const v = looksLikeLoginWall({ requestedUrl: IG_REQUESTED, finalUrl: IG_FINAL }); + expect(v.loginRequired).toBe(true); + expect(v.reason).toMatch(/accounts\/login/); + expect(v.reason).toMatch(/return parameter/); + }); + + it("catches it from the rendered password field alone, without the redirect", () => { + const v = looksLikeLoginWall({ + requestedUrl: IG_REQUESTED, + finalUrl: IG_REQUESTED, + html: '
', + }); + expect(v.loginRequired).toBe(true); + expect(v.reason).toMatch(/password field/); + }); + + it("recognises other common login routes", () => { + for (const path of [ + "/login", + "/signin", + "/sign-in", + "/users/sign_in", + "/session/new", + "/auth/login", + ]) { + const v = looksLikeLoginWall({ + requestedUrl: "https://dir.test/browse", + finalUrl: `https://dir.test${path}?return_to=%2Fbrowse`, + }); + expect(v.loginRequired, path).toBe(true); + } + }); + + it("does not misfire on a directory with a 'Log in' link in the header", () => { + const v = looksLikeLoginWall({ + requestedUrl: "https://dir.test/browse", + finalUrl: "https://dir.test/browse", + html: '
Log in
', + }); + expect(v.loginRequired).toBe(false); + }); + + it("does not misfire when the user deliberately seeded a login page", () => { + const v = looksLikeLoginWall({ + requestedUrl: "https://dir.test/login", + finalUrl: "https://dir.test/login", + html: '', + }); + expect(v.loginRequired).toBe(false); + }); + + it("does not treat an ordinary redirect as a login wall", () => { + const v = looksLikeLoginWall({ + requestedUrl: "https://dir.test/browse", + finalUrl: "https://dir.test/browse/page-1", + html: "", + }); + expect(v.loginRequired).toBe(false); + }); + + it("does not treat a bot challenge as a login wall", () => { + const v = looksLikeLoginWall({ + requestedUrl: "https://cf.test/search", + finalUrl: "https://cf.test/search", + html: "Just a moment...
", + }); + expect(v.loginRequired).toBe(false); + }); + + it("catches a cross-host SSO bounce", () => { + const v = looksLikeLoginWall({ + requestedUrl: "https://dir.test/browse", + finalUrl: "https://accounts.sso.test/signin?continue=https%3A%2F%2Fdir.test%2Fbrowse", + }); + expect(v.loginRequired).toBe(true); + }); + + it("survives malformed URLs without throwing", () => { + expect(looksLikeLoginWall({ requestedUrl: "not a url" }).loginRequired).toBe(false); + }); +});