From 6fed9a7683fe2715260ac5919c4b8144ab5e9b64 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Tue, 28 Jul 2026 08:56:58 +0000 Subject: [PATCH] feat(leads): step through paginated directories MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A listing page is a fraction of the directory behind it, and a fraction reported as a total is indistinguishable from a small directory. Seeds now walk the pager before opening anything. Following the next page by URL is tried before anything else, so the fetch-first path still applies and an ordinary paginated directory costs one cheap request per page rather than one browser render per page. The three ways of finding that URL are ordered by how much the page is actually asserting. rel="next" is the site stating it outright. Link text is a convention, and is matched whole — a link reading "next" is a pager, one reading "next steps" is prose — while an aria-label is matched on the word, because labels are written to be read aloud and say "Next page". Incrementing a page parameter is a guess, so it comes last and only when the URL already carries one; inventing ?page=2 asks every site on the internet for a page that may not exist. Disabled controls are refused, since a greyed-out arrow means this is the last page and following it loops. The guess needed a second guard that only a live run revealed. A site that ignores its page parameter answers every guess with the same page, and the walk dutifully fetched twenty-one identical pages before stopping at the cap. A page that contributes no new entries is now the end of the list whatever its pager claims, which took the same run from twenty-one page loads to two. Co-Authored-By: Claude Opus 5 (1M context) --- lib/outreach/discover.ts | 61 ++++++++++++++++++-- lib/outreach/pagination.ts | 113 +++++++++++++++++++++++++++++++++++++ tests/pagination.test.ts | 105 ++++++++++++++++++++++++++++++++++ 3 files changed, 274 insertions(+), 5 deletions(-) create mode 100644 lib/outreach/pagination.ts create mode 100644 tests/pagination.test.ts diff --git a/lib/outreach/discover.ts b/lib/outreach/discover.ts index 8c5a6e25..94f4a0a6 100644 --- a/lib/outreach/discover.ts +++ b/lib/outreach/discover.ts @@ -25,6 +25,7 @@ import { loadSeedCredential, makeSeedCodeWaiter, recordSeedCredentialResult, see import type { CodeWaiter } from "@/lib/sp/verificationChallenge"; import { looksLikeLoginWall } from "./loginWall"; import { extractPerson, type ExtractedPerson } from "./person"; +import { findNextPageUrl } from "./pagination"; import type { SeedCredentials } from "./seedLogin"; /** A person found on a page, with where they were found. */ @@ -263,6 +264,14 @@ const NON_DETAIL_PATH_RE = /** Cap on community pages opened per tick, since each is a full page load. */ const MAX_MINED_SOURCES = 5; +/** + * Listing pages walked per seed. + * + * A directory that paginates forever would otherwise be walked forever, and + * the page after the last one is usually the first one again. + */ +const MAX_LISTING_PAGES = 20; + const SEED_UA = "CrawlProofOutreach/1.0 (+https://crawlproof.com)"; /** Statuses that mean "the server refused a bot", not "the page is missing". */ @@ -427,11 +436,53 @@ export async function discoverFromSeed(input: { // at all on the first hop, which is exactly the signal to go deeper. const firstHopThin = merged.size < (input.detailHopThreshold ?? 3); if ((input.depth ?? 1) >= 2 && firstHopThin && merged.size < limit) { - const detailPages = extractSameHostLinks({ - html: seed.html, - sourceUrl: input.seedUrl, - limit: input.maxDetailPages ?? 12, - }); + // Walk the listing before opening anything. One page of a directory is + // a fraction of it, and a fraction reported as a total is + // indistinguishable from a small directory. + const wantDetails = input.maxDetailPages ?? 12; + const detailPages: string[] = []; + const seenPages = new Set([input.seedUrl]); + let pageHtml = seed.html; + let pageUrl: string | null = input.seedUrl; + + for (let page = 0; page < MAX_LISTING_PAGES && detailPages.length < wantDetails; page++) { + let addedThisPage = 0; + for (const link of extractSameHostLinks({ + html: pageHtml, + sourceUrl: pageUrl, + limit: wantDetails, + })) { + if (detailPages.length >= wantDetails) break; + if (!detailPages.includes(link)) { + detailPages.push(link); + addedThisPage += 1; + } + } + + // A page that contributes nothing new is the end of the list, whatever + // its pager claims. Incrementing a page parameter is a guess, and a + // site that ignores the parameter answers every guess with the same + // page — which walked twenty-one identical pages before this check. + if (page > 0 && addedThisPage === 0) break; + + const next: string | null = findNextPageUrl(pageHtml, pageUrl); + // A pager that points back at somewhere already walked is the end of + // the list, however it phrases itself. + if (!next || seenPages.has(next) || detailPages.length >= wantDetails) break; + seenPages.add(next); + + const loaded = await loadSeedHtml( + next, + allowRender, + input.credentials, + input.codeWaiter, + true, + ); + if ("error" in loaded) break; + pageHtml = loaded.html; + pageUrl = next; + } + if (seenPages.size > 1) notes.push(`walked ${seenPages.size} listing pages`); if (detailPages.length === 0) { notes.push("no listing entries found to open for a second hop"); } diff --git a/lib/outreach/pagination.ts b/lib/outreach/pagination.ts new file mode 100644 index 00000000..d1da83e0 --- /dev/null +++ b/lib/outreach/pagination.ts @@ -0,0 +1,113 @@ +// Stepping through a paginated directory. +// +// A listing page shows a fraction of what it holds — the run that prompted +// this saw sixteen of a directory's entries and reported that as the whole +// thing, which is indistinguishable from a short directory. Paging is what +// turns one page of a source into the source. +// +// Finding the next page by URL is deliberately tried before anything else. +// A resolvable href works on the fetch-first path, so an ordinary paginated +// directory costs one cheap request per page instead of one browser render +// per page. Clicking is the fallback for listings whose control carries no +// href at all, and it is the expensive one. + +// Visible text is matched whole: a link reading "next" is a pagination +// control, whereas one reading "next steps" is prose that happens to start +// with the word. +const NEXT_TEXT_RE = /^\s*(next|older|more|load more|show more|»|›|→|>>?)\s*$/i; + +// An aria-label is written to be read aloud — "Next page", "Go to next +// results" — so it is matched on the word rather than the whole string. +// Labels are short and purposeful, which makes this safe here and unsafe +// for link text. +const NEXT_LABEL_RE = /\bnext\b|\bload more\b|\bshow more\b/i; + +/** Params a site uses to mean "which page". */ +const PAGE_PARAMS = ["page", "p", "pg", "offset", "start", "from"]; + +function absolute(href: string, base: string): string | null { + try { + const url = new URL(href, base); + return url.protocol === "https:" || url.protocol === "http:" ? url.toString() : null; + } catch { + return null; + } +} + +/** + * The URL of the next listing page, or null. + * + * Ordered by how much the page is actually asserting. `rel="next"` is the + * site stating it outright; link text is a convention; incrementing a page + * parameter is a guess, and a guess that can loop, so it is last and only + * taken when the current URL already carries such a parameter — inventing + * `?page=2` on a URL that never had one invents a page that may not exist. + */ +export function findNextPageUrl(html: string, currentUrl: string): string | null { + // 1. / Last`; + expect(findNextPageUrl(html, BASE)).toBe("https://dir.test/browse?page=2"); + }); + + it("reads rel=next with the attributes in either order", () => { + const html = ``; + expect(findNextPageUrl(html, BASE)).toBe("https://dir.test/browse?page=3"); + }); + + it("follows a link whose text is a next control", () => { + const html = `Next`; + expect(findNextPageUrl(html, BASE)).toBe("https://dir.test/browse?page=2"); + }); + + it("follows an arrow-only control via aria-label", () => { + // The visible text is an icon, so the label is the only signal. + const html = ``; + expect(findNextPageUrl(html, BASE)).toBe("https://dir.test/browse?page=4"); + }); + + it("does not treat prose starting with 'next' as a control", () => { + // "Next steps" is a heading link, not pagination. Visible text is + // matched whole for exactly this reason. + expect(findNextPageUrl(`Next steps`, BASE)).toBeNull(); + }); + + it("refuses a disabled next control", () => { + // A disabled arrow means this is the last page; following it loops. + const html = `Next`; + expect(findNextPageUrl(html, BASE)).toBeNull(); + }); + + it("ignores an anchor to a fragment", () => { + expect(findNextPageUrl(`Next`, BASE)).toBeNull(); + }); + + it("never returns the page it was given", () => { + const html = ``; + expect(findNextPageUrl(html, BASE)).toBeNull(); + }); + + it("increments a page parameter the URL already carries", () => { + expect(findNextPageUrl("

no links

", "https://dir.test/browse?page=2")).toBe( + "https://dir.test/browse?page=3", + ); + }); + + it("does not invent a page parameter that was never there", () => { + // Fabricating ?page=2 asks for a page that may not exist, on every site. + expect(findNextPageUrl("

no links

", BASE)).toBeNull(); + }); + + it("leaves row-offset parameters alone", () => { + // offset counts rows, and the page size is not knowable from here — + // incrementing by one would re-fetch almost the same list. + expect(findNextPageUrl("

x

", "https://dir.test/browse?offset=20")).toBeNull(); + }); + + it("preserves the other query parameters when advancing", () => { + const next = findNextPageUrl("

x

", "https://dir.test/browse?skills=abc&page=1"); + expect(next).toContain("skills=abc"); + expect(next).toContain("page=2"); + }); + + it("resolves a relative href against the current page", () => { + expect(findNextPageUrl(``, "https://dir.test/browse/")).toBe( + "https://dir.test/browse/page/2", + ); + }); + + it("refuses a non-http scheme", () => { + expect(findNextPageUrl(``, BASE)).toBeNull(); + }); +}); + +describe("nextClickSelector", () => { + it("offers a selector when the control has no href", () => { + expect(nextClickSelector(``)).toBeTruthy(); + }); + + it("offers nothing when there is no such control", () => { + // Clicking needs a live browser per page, so it is only worth proposing + // when the page really has no followable link. + expect(nextClickSelector(`Next`)).toBeNull(); + }); +}); + +describe("walking a site that ignores its page parameter", () => { + it("still offers a next URL — the guard against looping is the caller's", () => { + // findNextPageUrl cannot know whether ?page=2 is real; only fetching it + // and seeing nothing new can. discoverFromSeed stops when a page adds no + // entries, which is what keeps a site that ignores the parameter from + // being walked to the page cap. + expect(findNextPageUrl("

no pager

", "https://dir.test/browse?page=1")).toBe( + "https://dir.test/browse?page=2", + ); + }); +});