Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
61 changes: 56 additions & 5 deletions lib/outreach/discover.ts
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,7 @@ import { loadSeedCredential, makeSeedCodeWaiter, recordSeedCredentialResult, see
import type { CodeWaiter } from "@/lib/sp/verificationChallenge";
import { looksLikeLoginWall } from "./loginWall";
import { extractPerson, type ExtractedPerson } from "./person";
import { findNextPageUrl } from "./pagination";
import type { SeedCredentials } from "./seedLogin";

/** A person found on a page, with where they were found. */
Expand Down Expand Up @@ -263,6 +264,14 @@ const NON_DETAIL_PATH_RE =
/** Cap on community pages opened per tick, since each is a full page load. */
const MAX_MINED_SOURCES = 5;

/**
* Listing pages walked per seed.
*
* A directory that paginates forever would otherwise be walked forever, and
* the page after the last one is usually the first one again.
*/
const MAX_LISTING_PAGES = 20;

const SEED_UA = "CrawlProofOutreach/1.0 (+https://crawlproof.com)";

/** Statuses that mean "the server refused a bot", not "the page is missing". */
Expand Down Expand Up @@ -427,11 +436,53 @@ export async function discoverFromSeed(input: {
// at all on the first hop, which is exactly the signal to go deeper.
const firstHopThin = merged.size < (input.detailHopThreshold ?? 3);
if ((input.depth ?? 1) >= 2 && firstHopThin && merged.size < limit) {
const detailPages = extractSameHostLinks({
html: seed.html,
sourceUrl: input.seedUrl,
limit: input.maxDetailPages ?? 12,
});
// Walk the listing before opening anything. One page of a directory is
// a fraction of it, and a fraction reported as a total is
// indistinguishable from a small directory.
const wantDetails = input.maxDetailPages ?? 12;
const detailPages: string[] = [];
const seenPages = new Set<string>([input.seedUrl]);
let pageHtml = seed.html;
let pageUrl: string | null = input.seedUrl;

for (let page = 0; page < MAX_LISTING_PAGES && detailPages.length < wantDetails; page++) {
let addedThisPage = 0;
for (const link of extractSameHostLinks({
html: pageHtml,
sourceUrl: pageUrl,
limit: wantDetails,
})) {
if (detailPages.length >= wantDetails) break;
if (!detailPages.includes(link)) {
detailPages.push(link);
addedThisPage += 1;
}
}

// A page that contributes nothing new is the end of the list, whatever
// its pager claims. Incrementing a page parameter is a guess, and a
// site that ignores the parameter answers every guess with the same
// page — which walked twenty-one identical pages before this check.
if (page > 0 && addedThisPage === 0) break;

const next: string | null = findNextPageUrl(pageHtml, pageUrl);
// A pager that points back at somewhere already walked is the end of
// the list, however it phrases itself.
if (!next || seenPages.has(next) || detailPages.length >= wantDetails) break;
seenPages.add(next);

const loaded = await loadSeedHtml(
next,
allowRender,
input.credentials,
input.codeWaiter,
true,
);
if ("error" in loaded) break;
pageHtml = loaded.html;
pageUrl = next;
}
if (seenPages.size > 1) notes.push(`walked ${seenPages.size} listing pages`);
if (detailPages.length === 0) {
notes.push("no listing entries found to open for a second hop");
}
Expand Down
113 changes: 113 additions & 0 deletions lib/outreach/pagination.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,113 @@
// Stepping through a paginated directory.
//
// A listing page shows a fraction of what it holds — the run that prompted
// this saw sixteen of a directory's entries and reported that as the whole
// thing, which is indistinguishable from a short directory. Paging is what
// turns one page of a source into the source.
//
// Finding the next page by URL is deliberately tried before anything else.
// A resolvable href works on the fetch-first path, so an ordinary paginated
// directory costs one cheap request per page instead of one browser render
// per page. Clicking is the fallback for listings whose control carries no
// href at all, and it is the expensive one.

// Visible text is matched whole: a link reading "next" is a pagination
// control, whereas one reading "next steps" is prose that happens to start
// with the word.
const NEXT_TEXT_RE = /^\s*(next|older|more|load more|show more|»|›|→|>>?)\s*$/i;

// An aria-label is written to be read aloud — "Next page", "Go to next
// results" — so it is matched on the word rather than the whole string.
// Labels are short and purposeful, which makes this safe here and unsafe
// for link text.
const NEXT_LABEL_RE = /\bnext\b|\bload more\b|\bshow more\b/i;

/** Params a site uses to mean "which page". */
const PAGE_PARAMS = ["page", "p", "pg", "offset", "start", "from"];

function absolute(href: string, base: string): string | null {
try {
const url = new URL(href, base);
return url.protocol === "https:" || url.protocol === "http:" ? url.toString() : null;
} catch {
return null;
}
}

/**
* The URL of the next listing page, or null.
*
* Ordered by how much the page is actually asserting. `rel="next"` is the
* site stating it outright; link text is a convention; incrementing a page
* parameter is a guess, and a guess that can loop, so it is last and only
* taken when the current URL already carries such a parameter — inventing
* `?page=2` on a URL that never had one invents a page that may not exist.
*/
export function findNextPageUrl(html: string, currentUrl: string): string | null {
// 1. <link rel="next"> / <a rel="next">
const relNext =
html.match(/<(?:link|a)[^>]+rel=["'][^"']*\bnext\b[^"']*["'][^>]*href=["']([^"']+)["']/i) ??
html.match(/<(?:link|a)[^>]+href=["']([^"']+)["'][^>]*rel=["'][^"']*\bnext\b[^"']*["']/i);
if (relNext?.[1]) {
const url = absolute(relNext[1], currentUrl);
if (url && url !== currentUrl) return url;
}

// 2. An anchor whose visible text is a next-page control. aria-label is
// checked too, because the arrow is often an icon with no text node.
for (const m of html.matchAll(/<a\b([^>]*)>([\s\S]{0,80}?)<\/a>/gi)) {
const attrs = m[1];
const text = m[2].replace(/<[^>]+>/g, " ").replace(/&[a-z]+;/gi, " ").trim();
const aria = attrs.match(/aria-label=["']([^"']+)["']/i)?.[1] ?? "";
if (!NEXT_TEXT_RE.test(text) && !NEXT_LABEL_RE.test(aria)) continue;
// A disabled control is on the last page and must not be followed.
if (/\baria-disabled=["']true["']|\bdisabled\b/i.test(attrs)) continue;
const href = attrs.match(/href=["']([^"']+)["']/i)?.[1];
if (!href || href.startsWith("#")) continue;
const url = absolute(href, currentUrl);
if (url && url !== currentUrl) return url;
}

// 3. Increment an existing page parameter. Only when one is already
// present — otherwise this fabricates a second page for every site.
try {
const url = new URL(currentUrl);
for (const param of PAGE_PARAMS) {
const raw = url.searchParams.get(param);
if (raw === null) continue;
const n = Number(raw);
if (!Number.isInteger(n)) continue;
// offset/start count rows, not pages, and there is no way to know the
// page size from here — so only page-numbered params are advanced.
if (param === "offset" || param === "start" || param === "from") continue;
url.searchParams.set(param, String(n + 1));
return url.toString();
}
} catch {
// Not a URL we can reason about.
}

return null;
}

/**
* A CSS selector for a clickable next control that carries no usable href.
*
* Only worth reaching for once findNextPageUrl has failed, since clicking
* requires a live browser for every page.
*/
export function nextClickSelector(html: string): string | null {
const hasHrefless =
/<button\b[^>]*>(\s*(next|more|load more|show more|»|›|→)\s*)<\/button>/i.test(html) ||
/<(?:button|a)\b[^>]*aria-label=["'](next[^"']*)["']/i.test(html);
if (!hasHrefless) return null;
// Matched broadly on purpose: the caller verifies the element is visible
// and enabled before clicking, so a selector that over-matches is cheap
// while a selector that under-matches loses the rest of the directory.
return [
'button:has-text("Next")',
'button:has-text("Load more")',
'button:has-text("Show more")',
'[aria-label*="Next" i]',
].join(", ");
}
105 changes: 105 additions & 0 deletions tests/pagination.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,105 @@
import { describe, it, expect } from "vitest";
import { findNextPageUrl, nextClickSelector } from "@/lib/outreach/pagination";

const BASE = "https://dir.test/browse";

describe("findNextPageUrl", () => {
it("prefers rel=next, which is the site saying so outright", () => {
const html = `<link rel="next" href="/browse?page=2"><a href="/browse?page=9">Last</a>`;
expect(findNextPageUrl(html, BASE)).toBe("https://dir.test/browse?page=2");
});

it("reads rel=next with the attributes in either order", () => {
const html = `<a href="/browse?page=3" rel="next">go</a>`;
expect(findNextPageUrl(html, BASE)).toBe("https://dir.test/browse?page=3");
});

it("follows a link whose text is a next control", () => {
const html = `<a href="/browse?page=2">Next</a>`;
expect(findNextPageUrl(html, BASE)).toBe("https://dir.test/browse?page=2");
});

it("follows an arrow-only control via aria-label", () => {
// The visible text is an icon, so the label is the only signal.
const html = `<a href="/browse?page=4" aria-label="Next page"><svg/></a>`;
expect(findNextPageUrl(html, BASE)).toBe("https://dir.test/browse?page=4");
});

it("does not treat prose starting with 'next' as a control", () => {
// "Next steps" is a heading link, not pagination. Visible text is
// matched whole for exactly this reason.
expect(findNextPageUrl(`<a href="/guide">Next steps</a>`, BASE)).toBeNull();
});

it("refuses a disabled next control", () => {
// A disabled arrow means this is the last page; following it loops.
const html = `<a href="/browse?page=2" aria-disabled="true">Next</a>`;
expect(findNextPageUrl(html, BASE)).toBeNull();
});

it("ignores an anchor to a fragment", () => {
expect(findNextPageUrl(`<a href="#next">Next</a>`, BASE)).toBeNull();
});

it("never returns the page it was given", () => {
const html = `<a href="/browse" rel="next">Next</a>`;
expect(findNextPageUrl(html, BASE)).toBeNull();
});

it("increments a page parameter the URL already carries", () => {
expect(findNextPageUrl("<p>no links</p>", "https://dir.test/browse?page=2")).toBe(
"https://dir.test/browse?page=3",
);
});

it("does not invent a page parameter that was never there", () => {
// Fabricating ?page=2 asks for a page that may not exist, on every site.
expect(findNextPageUrl("<p>no links</p>", BASE)).toBeNull();
});

it("leaves row-offset parameters alone", () => {
// offset counts rows, and the page size is not knowable from here —
// incrementing by one would re-fetch almost the same list.
expect(findNextPageUrl("<p>x</p>", "https://dir.test/browse?offset=20")).toBeNull();
});

it("preserves the other query parameters when advancing", () => {
const next = findNextPageUrl("<p>x</p>", "https://dir.test/browse?skills=abc&page=1");
expect(next).toContain("skills=abc");
expect(next).toContain("page=2");
});

it("resolves a relative href against the current page", () => {
expect(findNextPageUrl(`<a href="page/2" rel="next">Next</a>`, "https://dir.test/browse/")).toBe(
"https://dir.test/browse/page/2",
);
});

it("refuses a non-http scheme", () => {
expect(findNextPageUrl(`<a href="javascript:void(0)" rel="next">Next</a>`, BASE)).toBeNull();
});
});

describe("nextClickSelector", () => {
it("offers a selector when the control has no href", () => {
expect(nextClickSelector(`<button>Load more</button>`)).toBeTruthy();
});

it("offers nothing when there is no such control", () => {
// Clicking needs a live browser per page, so it is only worth proposing
// when the page really has no followable link.
expect(nextClickSelector(`<a href="/browse?page=2">Next</a>`)).toBeNull();
});
});

describe("walking a site that ignores its page parameter", () => {
it("still offers a next URL — the guard against looping is the caller's", () => {
// findNextPageUrl cannot know whether ?page=2 is real; only fetching it
// and seeing nothing new can. discoverFromSeed stops when a page adds no
// entries, which is what keeps a site that ignores the parameter from
// being walked to the page cap.
expect(findNextPageUrl("<p>no pager</p>", "https://dir.test/browse?page=1")).toBe(
"https://dir.test/browse?page=2",
);
});
});
Loading