From ec033b5a410513f60e3de04d80813208d5415aae Mon Sep 17 00:00:00 2001 From: Jan Buchar Date: Tue, 18 Aug 2026 14:48:18 +0200 Subject: [PATCH 01/16] Handle members that can be marked private/internal without additional changes `@internal` is only applied where the member genuinely is not promised surface. Overrides of a documented extension point are deliberately left untagged: the concrete `_launch`/`addProxyToLaunchOptions`/`isChromiumBasedBrowser` and `_close`/`_kill`/`_newPage`/`_getCookies`/`_setCookies` implementations restate a contract that `upgrading_v4.md` tells custom plugins and controllers to override, so tagging them would only shorten the report while claiming something false. `normalizeProxyOptions` and the `useRemoteConnection` overrides stay tagged -- their base declarations are internal too. --- .../examples/puppeteer_capture_screenshot.mdx | 2 +- .../puppeteer_crawler_utils_snapshot.ts | 4 +- docs/guides/configuration.mdx | 4 +- docs/public-api/crawlee-basic.api.md | 153 +--------- docs/public-api/crawlee-browser-pool.api.md | 120 +------- docs/public-api/crawlee-browser.api.md | 27 +- docs/public-api/crawlee-core.api.md | 204 ++----------- docs/public-api/crawlee-fs-storage.api.md | 14 - .../crawlee-got-scraping-client.api.md | 2 +- docs/public-api/crawlee-http-client.api.md | 12 +- docs/public-api/crawlee-http.api.md | 31 +- docs/public-api/crawlee-impit-client.api.md | 2 +- docs/public-api/crawlee-playwright.api.md | 28 +- docs/public-api/crawlee-puppeteer.api.md | 50 +--- docs/public-api/crawlee-stagehand.api.md | 23 +- docs/public-api/crawlee-types.api.md | 36 --- docs/public-api/crawlee-utils.api.md | 42 --- docs/public-api/crawlee.api.md | 22 -- docs/upgrading/upgrading_v4.md | 249 +++++++++++++++- .../autoscaling/concurrency_system.ts | 13 +- .../src/internals/basic-crawler.ts | 39 +-- .../src/internals/cookie_utils.ts | 5 - .../src/internals/crawlers/crawler_commons.ts | 9 +- .../internals/crawlers/error_snapshotter.ts | 30 +- .../src/internals/crawlers/error_tracker.ts | 44 +-- .../src/internals/crawlers/index.ts | 1 - .../internals/enqueue_links/enqueue_links.ts | 2 +- .../src/internals/enqueue_links/index.ts | 24 +- .../src/internals/enqueue_links/shared.ts | 15 +- .../basic-crawler/src/internals/router.ts | 2 + .../src/internals/session_pool/consts.ts | 4 +- .../src/internals/session_pool/session.ts | 2 + .../internals/session_pool/session_pool.ts | 19 +- .../src/internals/sitemap_request_loader.ts | 7 +- .../internals/throttling_request_manager.ts | 11 +- .../src/internals/browser-crawler.ts | 36 +-- .../src/internals/browser-launcher.ts | 4 +- .../abstract-classes/browser-controller.ts | 29 +- .../src/abstract-classes/browser-plugin.ts | 14 +- packages/browser-pool/src/anonymize-proxy.ts | 1 + packages/browser-pool/src/browser-pool.ts | 154 +++++----- .../src/container-proxy-server.ts | 51 ---- packages/browser-pool/src/events.ts | 1 - .../browser-pool/src/fingerprinting/types.ts | 39 +-- packages/browser-pool/src/index.ts | 7 +- packages/browser-pool/src/launch-context.ts | 1 + .../src/playwright/playwright-browser.ts | 1 + .../src/playwright/playwright-controller.ts | 1 + .../src/playwright/playwright-plugin.ts | 2 + .../src/puppeteer/puppeteer-controller.ts | 1 + .../src/puppeteer/puppeteer-plugin.ts | 6 +- .../browser-pool/src/remote-browser-pool.ts | 12 +- .../test/multiple-plugins.test.ts | 15 +- .../src/internals/cheerio-crawler.ts | 5 - .../core/src/memory-storage/memory-storage.ts | 2 +- packages/core/src/proxy_configuration.ts | 25 +- packages/core/src/request.ts | 2 + packages/core/src/service_locator.ts | 3 + packages/core/src/storages/dataset.ts | 11 +- packages/core/src/storages/index.ts | 12 +- packages/core/src/storages/key_value_store.ts | 1 + packages/core/src/storages/request_list.ts | 11 +- packages/core/src/storages/request_queue.ts | 1 + .../src/storages/storage_instance_manager.ts | 14 +- packages/core/src/storages/transaction.ts | 23 +- packages/crawlee/src/index.ts | 17 -- .../fs-storage/src/file-system-storage.ts | 54 ++-- .../test/default-storage-layout.test.ts | 23 +- .../test/key-value-store/adoption.test.ts | 13 +- .../request-queue-access.test.ts | 27 +- packages/fs-storage/test/storage-layout.ts | 15 + packages/got-scraping-client/README.md | 2 +- packages/got-scraping-client/src/index.ts | 14 +- packages/http-client/src/base-http-client.ts | 25 +- packages/http-client/src/fetch-http-client.ts | 2 +- packages/http-client/src/index.ts | 2 +- packages/http-client/src/response.ts | 8 +- .../src/internals/file-download.ts | 97 +------ .../src/internals/http-crawler.ts | 53 ++-- packages/impit-client/src/index.ts | 10 +- packages/playwright-crawler/src/index.ts | 3 +- .../internals/enqueue-links/click-elements.ts | 5 +- .../src/internals/playwright-crawler.ts | 70 ++--- .../src/internals/playwright-launcher.ts | 3 +- .../src/internals/utils/playwright-utils.ts | 22 +- .../utils/rendering-type-prediction.ts | 2 +- packages/puppeteer-crawler/src/index.ts | 12 +- .../internals/enqueue-links/click-elements.ts | 13 +- .../src/internals/puppeteer-crawler.ts | 13 +- .../src/internals/utils/puppeteer_utils.ts | 125 +------- packages/stagehand-crawler/src/index.ts | 5 - .../src/internals/stagehand-crawler.ts | 20 +- .../src/internals/stagehand-launcher.ts | 27 +- .../src/internals/stagehand-plugin.ts | 25 +- packages/types/src/browser.ts | 6 - packages/types/src/http-client.ts | 22 -- packages/utils/src/index.ts | 5 +- packages/utils/src/internal.ts | 3 +- packages/utils/src/internals/social.ts | 2 +- packages/utils/src/internals/validation.ts | 22 ++ test/browser-pool/browser-pool.test.ts | 267 ++++++++++++------ test/core/autoscaling/autoscaled_pool.test.ts | 26 +- .../autoscaling/concurrency_system.test.ts | 15 +- .../playwright_launcher.test.ts | 11 +- test/core/crawlers/basic_browser_crawler.ts | 4 +- test/core/crawlers/basic_crawler.test.ts | 11 +- test/core/crawlers/browser_crawler.test.ts | 9 +- test/core/crawlers/playwright_crawler.test.ts | 10 +- .../crawlers/rendering_type_predictor.test.ts | 2 +- .../core/enqueue_links/click_elements.test.ts | 15 +- test/core/error_snapshotter.test.ts | 67 +++-- test/core/got_scraping_http_client.test.ts | 4 +- test/core/impit_http_client.test.ts | 16 +- test/core/playwright_utils.test.ts | 3 +- .../puppeteer_request_interception.test.ts | 8 +- test/core/puppeteer_utils.test.ts | 91 +----- test/core/storages/dataset.test.ts | 9 + test/core/storages/storage_purge.test.ts | 12 +- .../core/storages/storage_transaction.test.ts | 15 +- 119 files changed, 1224 insertions(+), 1875 deletions(-) delete mode 100644 packages/browser-pool/src/container-proxy-server.ts create mode 100644 packages/fs-storage/test/storage-layout.ts diff --git a/docs/examples/puppeteer_capture_screenshot.mdx b/docs/examples/puppeteer_capture_screenshot.mdx index 7cc302f505fb..57de3aa9b361 100644 --- a/docs/examples/puppeteer_capture_screenshot.mdx +++ b/docs/examples/puppeteer_capture_screenshot.mdx @@ -36,7 +36,7 @@ Using `page.screenshot()`: -Using `utils.puppeteer.saveSnapshot()`: +Using `puppeteerUtils.saveSnapshot()`: {PuppeteerCrawlerUtilsSnapshotSource} diff --git a/docs/examples/puppeteer_crawler_utils_snapshot.ts b/docs/examples/puppeteer_crawler_utils_snapshot.ts index 13f9150911f8..81240989ad2c 100644 --- a/docs/examples/puppeteer_crawler_utils_snapshot.ts +++ b/docs/examples/puppeteer_crawler_utils_snapshot.ts @@ -1,4 +1,4 @@ -import { launchPuppeteer, utils } from 'crawlee'; +import { launchPuppeteer, puppeteerUtils } from 'crawlee'; const url = 'http://www.example.com/'; // Start a browser @@ -11,7 +11,7 @@ const page = await browser.newPage(); await page.goto(url); // Capture the screenshot -await utils.puppeteer.saveSnapshot(page, { key: 'my-key', saveHtml: false }); +await puppeteerUtils.saveSnapshot(page, { key: 'my-key', saveHtml: false }); // Close Puppeteer await browser.close(); diff --git a/docs/guides/configuration.mdx b/docs/guides/configuration.mdx index 68cca7501474..4067bc58b510 100644 --- a/docs/guides/configuration.mdx +++ b/docs/guides/configuration.mdx @@ -105,8 +105,8 @@ are launched in headful mode, i.e. with windows. Specifies the minimum log level, which can be one of the following values (in order of severity): `DEBUG`, `INFO`, `WARNING`, `ERROR` and `OFF`. By default, the log level is set to `INFO`, -which means that `DEBUG` messages are not printed to console. See the `utils.log` -namespace for logging utilities. +which means that `DEBUG` messages are not printed to console. See the `log` +instance for logging utilities. #### `CRAWLEE_VERBOSE_LOG` diff --git a/docs/public-api/crawlee-basic.api.md b/docs/public-api/crawlee-basic.api.md index 6814720f8aa4..2dea2533e4cb 100644 --- a/docs/public-api/crawlee-basic.api.md +++ b/docs/public-api/crawlee-basic.api.md @@ -15,7 +15,7 @@ import { CriticalError } from '@crawlee/core'; import { Dataset } from '@crawlee/core'; import type { DatasetExportOptions } from '@crawlee/core'; import { Dictionary } from '@crawlee/types'; -import { EnqueueStrategy } from '@crawlee/utils'; +import { EnqueueStrategy } from '@crawlee/utils/internal'; import type { EnqueueStrategyOption } from '@crawlee/core'; import { EventManager } from '@crawlee/core'; import type { HttpRequestOptions } from '@crawlee/types'; @@ -27,7 +27,7 @@ import type { ISessionPool } from '@crawlee/types'; import { KeyValueStore } from '@crawlee/core'; import { NonRetryableError } from '@crawlee/core'; import { PacingSignal } from '@crawlee/core'; -import { ParseSitemapOptions } from '@crawlee/utils'; +import type { ParseSitemapOptions } from '@crawlee/utils'; import type { ProxyInfo } from '@crawlee/types'; import type { ReadonlyDeep } from 'type-fest'; import { Request as Request_2 } from '@crawlee/core'; @@ -38,7 +38,6 @@ import type { RequestQueueOperationInfo } from '@crawlee/core'; import type { RequestQueueOperationOptions } from '@crawlee/core'; import type { RequestsLike } from '@crawlee/core'; import type { RequestSourceStatus } from '@crawlee/core'; -import { RobotsTxtFile } from '@crawlee/utils'; import type { SendRequestOptions } from '@crawlee/types'; import type { SessionFingerprint } from '@crawlee/types'; import type { SetStatusMessageOptions } from '@crawlee/types'; @@ -47,11 +46,9 @@ import { Source } from '@crawlee/core'; import type { StandardSchemaV1 } from '@standard-schema/spec'; import type { StorageBackend } from '@crawlee/types'; import type { StorageIdentifier } from '@crawlee/core'; -import type { StorageOpenOptions } from '@crawlee/core'; import { StorageWritePolicy } from '@crawlee/core'; import type { SyncStateConversion } from '@crawlee/core'; import { SystemInfo } from '@crawlee/core'; -import { TimeoutError } from '@apify/timeout'; // @public (undocumented) export class BasicCrawler, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends object = {}> { @@ -61,15 +58,10 @@ export class BasicCrawler; addRequests(requests: ReadonlyDeep>, options?: CrawlerAddRequestsOptions): Promise; - get basicContextPipeline(): ContextPipeline<{ - request: Request_2; - }, CrawlingContext>; // (undocumented) protected blockedStatusCodes: Set; protected buildContextPipeline(): ContextPipeline; get concurrencySystem(): IConcurrencySystem | undefined; - // (undocumented) - get contextPipeline(): ContextPipeline; protected createDefaultConcurrencySystem(options: ConcurrencySystemOptions): ConcurrencySystem; destroy(): Promise; exportData(path: string, format?: 'json' | 'csv', options?: DatasetExportOptions): Promise; @@ -77,13 +69,11 @@ export class BasicCrawler): ReturnType; getDataset(identifier?: string | StorageIdentifier): Promise; - protected getMessageFromError(error: Error, forceStack?: boolean): string | TimeoutError | undefined; + protected getMessageFromError(error: Error, forceStack?: boolean): string; protected getNavigationTimeoutMillis(): number; getRequestManager(): Promise; // @deprecated (undocumented) getRequestQueue(): Promise; - // (undocumented) - protected getRobotsTxtFileForUrl(url: string): Promise; get hasFinishedBefore(): boolean; // (undocumented) protected readonly httpClient: BaseHttpClient; @@ -170,23 +160,7 @@ export interface BasicCrawlingContext } // @public (undocumented) -export const BLOCKED_STATUS_CODES: number[]; - -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -interface BrowserCrawlingContext { - // (undocumented) - saveSnapshot: (options: { - key: string; - }) => Promise; -} - -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -interface BrowserPage { - // (undocumented) - content: () => Promise; -} +export const BLOCKED_STATUS_CODES: readonly number[]; // @public export interface CalculatedStatistics { @@ -212,7 +186,6 @@ export class ConcurrencySystem implements IConcurrencySystem { // (undocumented) get currentConcurrency(): number; get desiredConcurrency(): number; - set desiredConcurrency(value: number); getCurrentStatus(): SystemInfo; hasCapacityForTask(_consumer?: ConcurrencyConsumer): boolean; get isRunning(): boolean; @@ -319,16 +292,6 @@ export function createBasicRouter>(routes?: RouterRoutes>): RouterHandler>; -// @public (undocumented) -export interface CreateContextOptions { - // (undocumented) - proxyInfo?: ProxyInfo; - // (undocumented) - request: Request_2; - // (undocumented) - session: ISession; -} - // @public export interface CreateSession { // (undocumented) @@ -381,37 +344,6 @@ export interface ErrnoException extends Error { // @public export type ErrorHandler = (inputs: BaseContext & Partial, error: Error) => Awaitable; -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -interface ErrorSnapshot { - // (undocumented) - htmlFileName?: string; - // (undocumented) - htmlFileUrl?: string; - // (undocumented) - screenshotFileName?: string; - // (undocumented) - screenshotFileUrl?: string; -} - -// @public -export class ErrorSnapshotter { - // (undocumented) - static readonly BASE_MESSAGE = "An error occurred"; - captureSnapshot(error: ErrnoException, context: CrawlingContext & SnapshottableProperties): Promise; - contextCaptureSnapshot(context: BrowserCrawlingContext, fileName: string): Promise; - generateFilename(error: ErrnoException): string; - // (undocumented) - static readonly MAX_ERROR_CHARACTERS = 30; - // (undocumented) - static readonly MAX_FILENAME_LENGTH = 250; - // (undocumented) - static readonly MAX_HASH_LENGTH = 30; - saveHTMLSnapshot(html: string, keyValueStore: Pick, fileName: string): Promise; - // (undocumented) - static readonly SNAPSHOT_PREFIX = "ERROR_SNAPSHOT"; -} - // @public export class ErrorTracker { constructor(options?: Partial); @@ -419,19 +351,13 @@ export class ErrorTracker { add(error: ErrnoException): void; addAsync(error: ErrnoException, context?: CrawlingContext): Promise; // (undocumented) - captureSnapshot(storage: Record, error: ErrnoException, context: CrawlingContext & SnapshottableProperties): Promise; - // (undocumented) - errorSnapshotter?: ErrorSnapshotter; - // (undocumented) getMostPopularErrors(count: number): [number, string[]][]; // (undocumented) getUniqueErrorCount(): number; // (undocumented) reset(): void; - // (undocumented) - result: Record; - // (undocumented) - total: number; + get result(): Record; + get total(): number; } // @public (undocumented) @@ -555,8 +481,8 @@ export type LabeledSource> = str label?: undefined; }))); -// @public (undocumented) -export type LoadedRequest = WithRequired; +// @public +export type LoadedRequest = R & Required>; // @public export interface LoadSignal { @@ -589,9 +515,6 @@ export interface LoadSnapshot { isOverloaded: boolean; } -// @public (undocumented) -export const MAX_POOL_SIZE = 1000; - // @public export class MemoryLoadSignal implements LoadSignal { constructor(options?: MemoryLoadSignalOptions); @@ -625,9 +548,6 @@ export class NavigationSkippedError extends NonRetryableError { // @public export function parseRetryAfterHeader(value?: string | null): number | null; -// @public (undocumented) -export const PERSIST_STATE_KEY = "CRAWLEE_SESSION_POOL_STATE"; - // @public export interface PersistenceOptions { enable?: boolean; @@ -654,9 +574,6 @@ export class RequestHandlerError extends Error { constructor(error: unknown, options?: ErrorOptions); } -// @public -export type RequestManagerOpener = (identifier?: string | StorageIdentifier | null, options?: StorageOpenOptions) => Promise; - // @public export class RequestThrottledError extends RetryRequestError { constructor(message?: string); @@ -673,20 +590,10 @@ export type RequireContextPipeline ContextPipeline; }; -// @public (undocumented) -export interface ResponseLike { - // (undocumented) - headers?: Record | (() => Record); - // (undocumented) - url?: string | (() => string); -} - // @public (undocumented) export interface RestrictedCrawlingContext { addRequests: (requestsLike: ReadonlyDeep<(string | Source)[]>, options?: ReadonlyDeep) => Promise; getKeyValueStore: (identifier?: string | StorageIdentifier) => Promise>; - // (undocumented) - id: string; log: CrawleeLogger; proxyInfo?: ProxyInfo; pushData(data: ReadonlyDeep[0]>, datasetIdentifier?: string | StorageIdentifier): Promise; @@ -717,8 +624,6 @@ export class Router(schemas: Schemas): RouterHandler>; getHandler(label?: string | symbol): (ctx: Context) => Awaitable; - getMaxTimeoutSecs(): number | undefined; - getTimeoutSecs(label?: string | symbol): number | undefined; use(middleware: (ctx: Context) => Awaitable): void; } @@ -806,10 +711,6 @@ export class Session implements ISession { readonly userData: Dictionary; } -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -const SESSION_REUSE_STRATEGIES: readonly ['random', 'round-robin', 'use-until-failure']; - // @public (undocumented) export interface SessionOptions { // (undocumented) @@ -821,8 +722,6 @@ export interface SessionOptions { expiresAt?: Date; fingerprint?: SessionFingerprint; id?: string; - // (undocumented) - log?: CrawleeLogger; maxAgeSecs?: number; maxErrorScore?: number; maxUsageCount?: number; @@ -864,7 +763,7 @@ export interface SessionPoolOptions { } // @public (undocumented) -export type SessionReuseStrategy = (typeof SESSION_REUSE_STRATEGIES)[number]; +export type SessionReuseStrategy = 'random' | 'round-robin' | 'use-until-failure'; // @public export class SitemapRequestLoader implements IRequestLoader { @@ -912,14 +811,6 @@ export type SkippedRequestCallback = (args: { reason: SkippedRequestReason; }) => Awaitable; -// @public (undocumented) -export interface SnapshotResult { - // (undocumented) - htmlFileName?: string; - // (undocumented) - screenshotFileName?: string; -} - // @public export class SnapshotStore { clear(): void; @@ -929,15 +820,6 @@ export class SnapshotStore { useSampleWindow(maxSampleWindowMillis: number): void; } -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -interface SnapshottableProperties { - // (undocumented) - body?: unknown; - // (undocumented) - page?: BrowserPage; -} - // @public export interface StatisticPersistedState extends Omit { // (undocumented) @@ -1097,8 +979,6 @@ export class ThrottlingRequestManager; addRequestsBatched(requests: RequestsLike, options?: AddRequestsBatchedOptions): Promise; checkReadiness(): Promise; - // (undocumented) - drop(): Promise; fetchNextRequest(): Promise | null>; // (undocumented) getHandledCount(): Promise; @@ -1106,7 +986,6 @@ export class ThrottlingRequestManager; // (undocumented) getTotalCount(): Promise; - get innerManager(): T | undefined; // (undocumented) markRequestAsHandled(request: Request_2): Promise; purge(): Promise; @@ -1127,7 +1006,6 @@ export interface ThrottlingRequestManagerOptions; throttleBy?: 'hostname' | 'registrableDomain'; } @@ -1162,19 +1040,6 @@ interface UrlConstraints { // @public export type UrlPatternInput = GlobInput | RegExpInput; -// @public (undocumented) -export interface UrlPatternObject { - // (undocumented) - glob?: string; - // (undocumented) - regexp?: RegExp; -} - -// @public (undocumented) -export type WithRequired = T & { - [P in K]-?: T[P]; -}; - export * from "@crawlee/core"; diff --git a/docs/public-api/crawlee-browser-pool.api.md b/docs/public-api/crawlee-browser-pool.api.md index 3938321c202d..11ab2356eba7 100644 --- a/docs/public-api/crawlee-browser-pool.api.md +++ b/docs/public-api/crawlee-browser-pool.api.md @@ -9,28 +9,19 @@ import type { BrowserContext } from 'playwright'; import type { BrowserFingerprintWithHeaders } from 'fingerprint-generator'; import type { BrowserType } from 'playwright'; import type { Cookie } from '@crawlee/types'; -import { CrawleeLogger } from '@crawlee/core'; import { CriticalError } from '@crawlee/core'; import type { Dictionary } from '@crawlee/types'; import { EventEmitter } from 'node:events'; -import { FingerprintGenerator as FingerprintGenerator_2 } from 'fingerprint-generator'; +import { FingerprintGenerator } from 'fingerprint-generator'; import type { FingerprintGeneratorOptions as FingerprintGeneratorOptions_2 } from 'fingerprint-generator'; -import { FingerprintInjector } from 'fingerprint-injector'; import { IBrowserPool } from '@crawlee/types'; import { NewPageOptions } from '@crawlee/types'; import type { Page } from 'playwright'; import type { PageState } from '@crawlee/types'; import type Puppeteer from 'puppeteer'; import type * as PuppeteerTypes from 'puppeteer'; -import QuickLRU from 'quick-lru'; import { TypedEmitter } from 'tiny-typed-emitter'; -// @public (undocumented) -export interface AnonymizeProxySugarOptions { - // (undocumented) - ignoreProxyCertificate?: boolean; -} - // @public (undocumented) export enum BROWSER_CONTROLLER_EVENTS { // (undocumented) @@ -39,8 +30,6 @@ export enum BROWSER_CONTROLLER_EVENTS { // @public (undocumented) export enum BROWSER_POOL_EVENTS { - // (undocumented) - BROWSER_CLOSED = "browserClosed", // (undocumented) BROWSER_LAUNCHED = "browserLaunched", // (undocumented) @@ -67,27 +56,17 @@ export abstract class BrowserController; // (undocumented) readonly id: string; - // (undocumented) - isActive: boolean; kill(): Promise; // (undocumented) protected abstract _kill(): Promise; - // (undocumented) - lastPageOpenedAt: number; launchContext: LaunchContext; // (undocumented) - protected readonly log: CrawleeLogger; - // (undocumented) protected abstract _newPage(pageOptions?: NewPageOptions): Promise; - // (undocumented) - abstract normalizeProxyOptions(proxyUrl: string | undefined, pageOptions: any): Record; proxyUrl?: string; // (undocumented) setCookies(page: NewPageResult, cookies: Cookie[]): Promise; // (undocumented) protected abstract _setCookies(page: NewPageResult, cookies: Cookie[]): Promise; - // (undocumented) - totalPages: number; waitForActive(): Promise; } @@ -114,15 +93,6 @@ export enum BrowserName { safari = "safari" } -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -interface BrowserOptions { - // (undocumented) - browserContext: BrowserContext; - // (undocumented) - version: string; -} - // @public export abstract class BrowserPlugin[0], LaunchResult extends CommonBrowser = UnwrapPromise>, NewPageOptions = Parameters[0], NewPageResult = UnwrapPromise>> { constructor(library: Library, options?: BrowserPluginOptions); @@ -146,8 +116,6 @@ export abstract class BrowserPlugin; constructor(options: Options & BrowserPoolHooks); // (undocumented) - activeBrowserControllers: Set; - // (undocumented) browserPlugins: BrowserPlugins; closeAllBrowsers(): Promise; - // (undocumented) - closeInactiveBrowserAfterMillis: number; closePage(page: PageReturn, options?: { error?: Error; }): Promise; destroy(): Promise; extractPageState(page: PageReturn): Promise; // (undocumented) - fingerprintCache?: QuickLRU; - // (undocumented) - fingerprintGenerator?: FingerprintGenerator_2; - // (undocumented) - fingerprintInjector?: FingerprintInjector; - // (undocumented) - fingerprintOptions: FingerprintOptions; + fingerprintGenerator?: FingerprintGenerator; getBrowserControllerByPage(page: PageReturn): BrowserControllerReturn | undefined; getPage(id: string): PageReturn | undefined; getPageId(page: PageReturn): string | undefined; - hasActiveBrowserWithFreeCapacity(): boolean; - hasFreeBrowserSlot(): boolean; injectPageState(page: PageReturn, state: PageState): Promise; - // (undocumented) - maxOpenBrowsers: number; - // (undocumented) - maxOpenPagesPerBrowser: number; newPage(options?: BrowserPoolNewPageOptions): Promise; newPageInNewBrowser(options?: BrowserPoolNewPageInNewBrowserOptions): Promise; newPageWithEachPlugin(optionsList?: Omit, 'browserPlugin'>[]): Promise; - // (undocumented) - operationTimeoutMillis: number; - // (undocumented) - pageCounter: number; - // (undocumented) - pageIds: WeakMap; - // (undocumented) - pages: Map; - // (undocumented) - pageToBrowserController: WeakMap; - // (undocumented) - postLaunchHooks: PostLaunchHook[]; - // (undocumented) - postPageCloseHooks: PostPageCloseHook[]; - // (undocumented) - postPageCreateHooks: PostPageCreateHook[]; - // (undocumented) - preLaunchHooks: PreLaunchHook[]; - // (undocumented) - prePageCloseHooks: PrePageCloseHook[]; - // (undocumented) - prePageCreateHooks: PrePageCreateHook[]; releaseAllBrowsers(): Promise; retireAllBrowsers(): void; - // (undocumented) - retireBrowserAfterPageCount: number; retireBrowserByPage(page: PageReturn): void; retireBrowserController(browserController: BrowserControllerReturn): void; - // (undocumented) - retiredBrowserControllers: Set; - // (undocumented) - startingBrowserControllers: Set; - // (undocumented) - useFingerprints?: boolean; } // @public (undocumented) @@ -294,14 +216,6 @@ export interface BrowserPoolOptions GetFingerprintReturn; -} - // @public (undocumented) export interface FingerprintGeneratorOptions extends Partial { } @@ -362,16 +270,6 @@ export interface FingerprintOptions { useFingerprintCache?: boolean; } -// @public (undocumented) -export interface GetFingerprintReturn { - // (undocumented) - fingerprint: BrowserFingerprintWithHeaders; -} - -// Not exported by the entry point; reachable only as a referenced type. -// @public -type HttpVersion = (typeof SUPPORTED_HTTP_VERSIONS)[number]; - // @public export interface IBrowserController { readonly browser: unknown; @@ -431,7 +329,6 @@ export interface LaunchContextOptions; id?: string; ignoreProxyCertificate?: boolean; - isRemote?: boolean; launchOptions: LibraryOptions; // (undocumented) proxyUrl?: string; @@ -457,7 +354,6 @@ export enum OperatingSystemsName { export class PlaywrightBrowser extends EventEmitter { // (undocumented) [Symbol.asyncDispose](): Promise; - constructor(options: BrowserOptions); // (undocumented) browserType(): BrowserType; // (undocumented) @@ -491,8 +387,6 @@ export class PlaywrightController extends BrowserController[0]): Promise; // (undocumented) - normalizeProxyOptions(proxyUrl: string | undefined, pageOptions: any): Record; - // (undocumented) protected _setCookies(page: Page, cookies: Cookie[]): Promise; } @@ -506,7 +400,6 @@ export class PlaywrightPlugin extends BrowserPlugin): Promise; - useRemoteConnection(connection: RemoteConnection, parameters?: RemoteConnectionParameters): void; } // @public @@ -538,8 +431,6 @@ export class PuppeteerController extends BrowserController; // (undocumented) - normalizeProxyOptions(proxyUrl: string | undefined, pageOptions: any): Record; - // (undocumented) protected _setCookies(page: PuppeteerTypes.Page, cookies: Cookie[]): Promise; } @@ -562,7 +453,6 @@ export class PuppeteerPlugin extends BrowserPlugin): boolean; // (undocumented) protected _launch(launchContext: LaunchContext): Promise; - useRemoteConnection(connection: RemoteConnection, parameters?: RemoteConnectionParameters): void; } // @public @@ -575,7 +465,6 @@ export class RemoteBrowserPool implements IBrowserPool { // (undocumented) [Symbol.asyncDispose](): Promise; constructor(options: RemoteBrowserPoolOptions); - readonly browserPool: BrowserPool; // (undocumented) closePage(page: Page, options?: { error?: Error; @@ -602,7 +491,6 @@ export interface RemoteBrowserPoolOptions { endpoint: string; context?: Record; }) => unknown; - slotPollIntervalMillis?: number; } // @public @@ -647,10 +535,6 @@ export interface ResolvedRemoteEndpoint { // @public type SafeParameters any> = unknown[] extends Parameters ? any : Parameters; -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -const SUPPORTED_HTTP_VERSIONS: readonly ['1', '2']; - // @public (undocumented) export type UnwrapPromise = T extends PromiseLike ? UnwrapPromise : T; diff --git a/docs/public-api/crawlee-browser.api.md b/docs/public-api/crawlee-browser.api.md index 140e0132ebc4..c1e094b66769 100644 --- a/docs/public-api/crawlee-browser.api.md +++ b/docs/public-api/crawlee-browser.api.md @@ -9,8 +9,6 @@ import type { Awaitable } from '@crawlee/types'; import { BasicCrawler } from '@crawlee/basic'; import { BasicCrawlerOptions } from '@crawlee/basic'; import type { BrowserPluginOptions } from '@crawlee/browser-pool'; -import type { BrowserPoolHooks } from '@crawlee/browser-pool'; -import type { BrowserPoolOptions } from '@crawlee/browser-pool'; import type { CommonPage } from '@crawlee/browser-pool'; import { ContextPipeline } from '@crawlee/basic'; import type { CrawlerRemoteBrowserOptions } from '@crawlee/browser-pool'; @@ -22,14 +20,10 @@ import type { ExtractLinksOptions } from '@crawlee/basic'; import type { GetUserDataFromRequest } from '@crawlee/basic'; import type { IBrowserPool } from '@crawlee/types'; import type { LoadedRequest } from '@crawlee/basic'; -import type { RemoteBrowserPoolOptions } from '@crawlee/browser-pool'; import { Request as Request_2 } from '@crawlee/basic'; import type { RequestHandler } from '@crawlee/basic'; import type { RouterHandler } from '@crawlee/basic'; -// @public -export function assertBrowserPoolNotConfigured(crawlerName: string, ignoredOptions: Dictionary): void; - // Not exported by the entry point; reachable only as a referenced type. // @public (undocumented) interface BaseResponse { @@ -39,27 +33,18 @@ interface BaseResponse { } // @public -export abstract class BrowserCrawler = BrowserCrawlingContext, ContextExtension = Dictionary, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends object = {}, GoToOptions extends Dictionary = Dictionary> extends BasicCrawler { - protected constructor(options: BrowserCrawlerOptions & { - contextPipelineBuilder: () => ContextPipeline; - browserPoolBuilder: (remoteBrowser?: CrawlerRemoteBrowserOptions) => OwnedBrowserPool; - }); +export abstract class BrowserCrawler = BrowserCrawlingContext, ContextExtension = Dictionary, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends object = {}, GoToOptions extends Dictionary = Dictionary> extends BasicCrawler { get browserPool(): IBrowserPool; // (undocumented) protected buildContextPipeline(): ContextPipeline>; // (undocumented) destroy(): Promise; // (undocumented) - protected getNavigationTimeoutMillis(): number; - // (undocumented) protected readonly ignoreIframes: boolean; // (undocumented) protected readonly ignoreShadowRoots: boolean; // (undocumented) - launchContext: BrowserLaunchContext; - // (undocumented) protected abstract navigationHandler(crawlingContext: BrowserCrawlingContext, gotoOptions: GoToOptions): Promise; - protected runRequestHandler(crawlingContext: ExtendedContext): Promise; teardown(): Promise; } @@ -70,8 +55,6 @@ export interface BrowserCrawlerOptions; ignoreIframes?: boolean; ignoreShadowRoots?: boolean; - // (undocumented) - launchContext?: BrowserLaunchContext; navigationTimeoutSecs?: number; postNavigationHooks?: BrowserHook[]; preNavigationHooks?: BrowserHook[]; @@ -106,14 +89,6 @@ export interface BrowserLaunchContext extends BrowserPluginO userDataDir?: string; } -// @public -export type LauncherBrowserPoolOptions = Omit & { - [Hook in keyof BrowserPoolHooks]?: readonly ((...args: any[]) => unknown)[]; -}; - -// @public -export type LauncherRemoteBrowserPoolOptions = Omit; - // @public export type OwnedBrowserPool = IBrowserPool & { releaseAllBrowsers: () => Promise; diff --git a/docs/public-api/crawlee-core.api.md b/docs/public-api/crawlee-core.api.md index 4d65b43a00bb..250a5e485a58 100644 --- a/docs/public-api/crawlee-core.api.md +++ b/docs/public-api/crawlee-core.api.md @@ -19,7 +19,6 @@ import type { DatasetInfo } from '@crawlee/types'; import { Dictionary } from '@crawlee/types'; import { EnqueueStrategy } from '@crawlee/utils'; import type { KeyValueStoreBackend } from '@crawlee/types'; -import type { KeyValueStoreInfo } from '@crawlee/types'; import type { LiteralUnion } from 'type-fest'; import { Log } from '@apify/log'; import log from '@apify/log'; @@ -74,14 +73,6 @@ export class ApifyLogAdapter extends BaseCrawleeLogger { export { ArgumentValidationError } -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -class BaseClient { - constructor(id: string); - // (undocumented) - id: string; -} - // @public export abstract class BaseCrawleeLogger implements CrawleeLogger { constructor(options?: Partial); @@ -180,8 +171,6 @@ export class CriticalError extends NonRetryableError { export class Dataset { [Symbol.asyncIterator](): AsyncGenerator; // (undocumented) - backend: DatasetBackend; - // (undocumented) readonly configuration: Configuration; drop(): Promise; entries(options?: DatasetIteratorOptions): AsyncIterable<[number, Data]> & Promise<[number, Data][]>; @@ -196,12 +185,10 @@ export class Dataset { static getData(options?: DatasetDataOptions): Promise>; getInfo(): Promise; // (undocumented) - id: string; - // (undocumented) - log: CrawleeLogger; + readonly id: string; map(iteratee: DatasetMapper, options?: DatasetIteratorOptions): Promise; // (undocumented) - name?: string; + readonly name?: string; static open(identifier?: string | StorageIdentifier | null, options?: StorageOpenOptions): Promise>; purge(): Promise; pushData(data: Data | Data[]): Promise; @@ -257,29 +244,11 @@ export interface DatasetExportToOptions extends DatasetExportOptions { export interface DatasetIteratorOptions extends Omit { } -// @public -export interface DatasetJournalEntry { - items: Dictionary[]; - // (undocumented) - recordedAt: Date; - // (undocumented) - storageId: string; - // (undocumented) - type: 'dataset'; -} - // @public export interface DatasetMapper { (item: Data, index: number): Awaitable; } -// @public (undocumented) -export interface DatasetOptions { - // (undocumented) - backend: DatasetBackend; - metadata: DatasetInfo; -} - // @public export interface DatasetReducer { // (undocumented) @@ -390,10 +359,6 @@ export type FieldsOutput> = { [K in keyof F]: z.output; }; -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -type Hashable = string; - // Not exported by the entry point; reachable only as a referenced type. // @public (undocumented) interface Intervals { @@ -405,7 +370,9 @@ interface Intervals { // @public export interface IProxyConfiguration { - newProxyInfo(options?: NewUrlOptions): Promise; + newProxyInfo(options?: { + request?: Request_2; + }): Promise; } // @public @@ -440,20 +407,6 @@ export interface IStorage { name?: string; } -// @public -export interface JournaledRequest { - // (undocumented) - label?: string; - snapshot?: Dictionary; - // (undocumented) - uniqueKey: string; - // (undocumented) - url: string; -} - -// @public (undocumented) -export type JournalEntry = DatasetJournalEntry | KeyValueStoreJournalEntry | RequestQueueJournalEntry; - // @public export interface KeyConsumer { // (undocumented) @@ -499,26 +452,6 @@ export interface KeyValueStoreIteratorOptions { prefix?: string; } -// @public -export interface KeyValueStoreJournalEntry { - // (undocumented) - key: string; - // (undocumented) - options?: RecordOptions; - // (undocumented) - storageId: string; - // (undocumented) - type: 'keyValueStore'; - value: unknown; -} - -// @public (undocumented) -export interface KeyValueStoreOptions { - // (undocumented) - backend: KeyValueStoreBackend; - metadata: KeyValueStoreInfo; -} - // @public export interface KeyValueStoreRawRecord { // (undocumented) @@ -583,7 +516,7 @@ export class MemoryStorageBackend implements storage.StorageBackend { // (undocumented) createKeyValueStoreBackend(options?: storage.StorageIdentifier): Promise; // (undocumented) - createRequestQueueBackend(options?: storage.StorageIdentifier): Promise; + createRequestQueueBackend(options?: storage.StorageIdentifier): Promise; getStorageBackendCacheKey(): string; // (undocumented) readonly logger?: CrawleeLogger; @@ -598,13 +531,6 @@ export interface MemoryStorageOptions { logger?: CrawleeLogger; } -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -interface NewUrlOptions { - // (undocumented) - request?: Request_2; -} - // @public export class NonRetryableError extends Error { } @@ -637,8 +563,12 @@ export class ProxyConfiguration implements IProxyConfiguration { constructor(options?: ProxyConfigurationOptions); // (undocumented) readonly isManInTheMiddle = false; - newProxyInfo(options?: NewUrlOptions): Promise; - newUrl(options?: NewUrlOptions): Promise; + newProxyInfo(options?: { + request?: Request_2; + }): Promise; + newUrl(options?: { + request?: Request_2; + }): Promise; } // @public (undocumented) @@ -652,7 +582,7 @@ export interface ProxyConfigurationFunction { // @public (undocumented) export interface ProxyConfigurationOptions { newUrlFunction?: ProxyConfigurationFunction; - proxyUrls?: UrlList; + proxyUrls?: (string | null)[]; } // Not exported by the entry point; reachable only as a referenced type. @@ -733,8 +663,6 @@ class Request_2 { set sessionId(value: string | undefined); get skipNavigation(): boolean; set skipNavigation(value: boolean); - get skippedReason(): SkippedRequestReason | undefined; - set skippedReason(value: SkippedRequestReason | undefined); get state(): RequestState; set state(value: RequestState); uniqueKey: string; @@ -758,7 +686,7 @@ export class RequestList implements IRequestLoader { getTotalCount(): Promise; // (undocumented) markRequestAsHandled(request: Request_2): Promise; - static open(listNameOrOptions: string | null | RequestListOptions, sources?: RequestListSource[], options?: RequestListOptions): Promise; + static open(listNameOrOptions: string | null | RequestListOptions, sources?: (string | Source)[], options?: RequestListOptions): Promise; // (undocumented) persistState(): Promise; teardown(): Promise; @@ -772,17 +700,13 @@ export interface RequestListOptions { persistRequestsKey?: string; persistStateKey?: string; proxyConfiguration?: IProxyConfiguration; - sources?: RequestListSource[]; + sources?: (string | Source)[]; sourcesFunction?: RequestListSourcesFunction; state?: RequestListState; } -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -type RequestListSource = string | Source; - // @public (undocumented) -export type RequestListSourcesFunction = () => Promise; +export type RequestListSourcesFunction = () => Promise<(string | Source)[]>; // @public export interface RequestListState { @@ -873,71 +797,6 @@ export class RequestQueue implements IStorage, IRequestManager { get stats(): RequestQueueStats; } -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -class RequestQueueBackend_2 extends BaseClient implements storage.RequestQueueBackend { - constructor(options: RequestQueueBackendOptions); - // (undocumented) - accessedAt: Date; - // (undocumented) - addBatchOfRequests(requests: storage.RequestSchema[], options?: storage.RequestQueueOperationOptions): Promise; - cacheKey: string; - // (undocumented) - createdAt: Date; - // (undocumented) - drop(): Promise; - // (undocumented) - fetchNextRequest(): Promise; - // (undocumented) - getMetadata(): Promise; - // (undocumented) - getRequest(uniqueKey: string): Promise; - // (undocumented) - handledRequestCount: number; - // (undocumented) - isEmpty(): Promise; - // (undocumented) - isFinished(): Promise; - listItems(): Promise; - // (undocumented) - markRequestAsHandled(request: storage.UpdateRequestSchema): Promise; - // (undocumented) - modifiedAt: Date; - // (undocumented) - name?: string; - // (undocumented) - pendingRequestCount: number; - // (undocumented) - purge(): Promise; - // (undocumented) - reclaimRequest(request: storage.UpdateRequestSchema, options?: storage.RequestQueueOperationOptions): Promise; - // (undocumented) - toRequestQueueInfo(): storage.RequestQueueInfo; -} - -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -interface RequestQueueBackendOptions { - cacheKey?: string; - // (undocumented) - id?: string; - // (undocumented) - name?: string; - // (undocumented) - storageBackend: MemoryStorageBackend; -} - -// @public -export interface RequestQueueJournalEntry { - // (undocumented) - forefront: boolean; - // (undocumented) - requests: JournaledRequest[]; - // (undocumented) - type: 'requestQueue'; - writeThrough: boolean; -} - // @public (undocumented) export interface RequestQueueOperationInfo extends QueueOperationInfo { // (undocumented) @@ -951,14 +810,6 @@ export interface RequestQueueOperationOptions { forefront?: boolean; } -// @public (undocumented) -export interface RequestQueueOptions { - // (undocumented) - backend: RequestQueueBackend; - metadata: RequestQueueInfo; - proxyConfiguration?: IProxyConfiguration; -} - // @public export interface RequestQueueStats { headItemReadCount: number; @@ -1013,9 +864,6 @@ export class RequestValidationError extends NonRetryableError { // @public (undocumented) export type ResolvedConfigValues = FieldsOutput; -// @public -export function resolveStorageIdentifier(identifier: string | StorageIdentifier | null | undefined, storageBackend: StorageBackend, storageType: 'Dataset' | 'KeyValueStore' | 'RequestQueue'): Promise; - // @public export interface SchemaIssue { // (undocumented) @@ -1051,10 +899,6 @@ export class ServiceLocator implements ServiceLocatorInterface { // (undocumented) getStorageBackend(): StorageBackend; // (undocumented) - getStorageInstanceManager(): StorageInstanceManager; - // (undocumented) - reset(): void; - // (undocumented) setConfiguration(configuration: Configuration): void; // (undocumented) setEventManager(eventManager: EventManager): void; @@ -1075,7 +919,6 @@ interface ServiceLocatorInterface { getEventManager(): EventManager; getLogger(): CrawleeLogger; getStorageBackend(): StorageBackend; - getStorageInstanceManager(): StorageInstanceManager; setConfiguration(configuration: Configuration): void; setEventManager(eventManager: EventManager): void; setLogger(logger: CrawleeLogger): void; @@ -1117,7 +960,7 @@ export class StorageInstanceManager { clearCache(): void; openStorage(cls: Constructor, input: (ExplicitStorageIdentifier | DefaultStorageIdentifier) & { backendOpener: () => Promise; - backendCacheKey: Hashable; + backendCacheKey: string; }): Promise; removeFromCache(instance: IStorage): void; } @@ -1130,13 +973,6 @@ export interface StorageOpenOptions { storageBackend?: StorageBackend; } -// @public -export class StorageStatsTracker> { - constructor(initial: T); - add(key: keyof T, by?: number): void; - get current(): T; -} - // @public export class StorageTransaction implements StorageTransactionView { afterCommit(callback: (error?: Error) => Awaitable): void; @@ -1153,13 +989,11 @@ export class StorageTransaction implements StorageTransactionView { label?: string; }[]; get isActive(): boolean; - readonly journal: JournalEntry[]; // (undocumented) get keyValueStoreChanges(): Record>; - readonly policy: StorageWritePolicy; rollback(): void; run(callback: () => Awaitable): Promise; // (undocumented) @@ -1222,10 +1056,6 @@ export interface SystemInfo { storageBackendInfo: LoadSignalInfo; } -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -type UrlList = (string | null)[]; - // @public export function useState(name?: string, defaultValue?: State, options?: UseStateOptions): Promise; diff --git a/docs/public-api/crawlee-fs-storage.api.md b/docs/public-api/crawlee-fs-storage.api.md index 3d11b20b8d58..1625b820dda3 100644 --- a/docs/public-api/crawlee-fs-storage.api.md +++ b/docs/public-api/crawlee-fs-storage.api.md @@ -4,7 +4,6 @@ ```ts -import type { CrawleeLogger } from '@crawlee/types'; import type * as storage from '@crawlee/types'; // @public @@ -16,21 +15,9 @@ export class FileSystemStorageBackend implements storage.StorageBackend { createKeyValueStoreBackend(options?: storage.StorageIdentifier): Promise; // (undocumented) createRequestQueueBackend(options?: storage.StorageIdentifier): Promise; - // (undocumented) - readonly datasetsDirectory: string; getStorageBackendCacheKey(): string; - // (undocumented) - readonly keyValueStoresDirectory: string; - // (undocumented) - readonly localDataDirectory: string; - // (undocumented) - readonly logger?: CrawleeLogger; purge(): Promise; // (undocumented) - readonly requestQueueAccess: 'single' | 'shared'; - // (undocumented) - readonly requestQueuesDirectory: string; - // (undocumented) storageExists(id: string, type: 'Dataset' | 'KeyValueStore' | 'RequestQueue'): Promise; teardown(): Promise; } @@ -39,7 +26,6 @@ export class FileSystemStorageBackend implements storage.StorageBackend { export interface FileSystemStorageOptions { inputKey?: string; localDataDirectory: string; - logger?: CrawleeLogger; requestQueueAccess?: 'single' | 'shared'; } diff --git a/docs/public-api/crawlee-got-scraping-client.api.md b/docs/public-api/crawlee-got-scraping-client.api.md index fb497938bf79..0301a1a577cc 100644 --- a/docs/public-api/crawlee-got-scraping-client.api.md +++ b/docs/public-api/crawlee-got-scraping-client.api.md @@ -10,7 +10,7 @@ import { CustomFetchOptions } from '@crawlee/http-client'; // @public export class GotScrapingHttpClient extends BaseHttpClient { // (undocumented) - fetch(request: Request, options?: RequestInit & CustomFetchOptions): Promise; + protected fetch(request: Request, options?: RequestInit & CustomFetchOptions): Promise; } // (No @packageDocumentation comment for this package) diff --git a/docs/public-api/crawlee-http-client.api.md b/docs/public-api/crawlee-http-client.api.md index ce9b5e432fd5..559f9827fc25 100644 --- a/docs/public-api/crawlee-http-client.api.md +++ b/docs/public-api/crawlee-http-client.api.md @@ -33,22 +33,16 @@ export class FetchHttpClient extends BaseHttpClient { logger?: CrawleeLogger; }); // (undocumented) - fetch(request: Request, options?: RequestInit & CustomFetchOptions): Promise; -} - -// @public (undocumented) -export interface IResponseWithUrl extends Response { - // (undocumented) - url: string; + protected fetch(request: Request, options?: RequestInit & CustomFetchOptions): Promise; } // @public -export class ResponseWithUrl extends Response implements IResponseWithUrl { +export class ResponseWithUrl extends Response { constructor(body: BodyInit | null, init: ResponseInit & { url?: string; }); // (undocumented) - url: string; + readonly url: string; } // (No @packageDocumentation comment for this package) diff --git a/docs/public-api/crawlee-http.api.md b/docs/public-api/crawlee-http.api.md index 1f930843a85d..786be7dbea3e 100644 --- a/docs/public-api/crawlee-http.api.md +++ b/docs/public-api/crawlee-http.api.md @@ -9,7 +9,6 @@ import type { Awaitable } from '@crawlee/types'; import { BasicCrawler } from '@crawlee/basic'; import { BasicCrawlerOptions } from '@crawlee/basic'; import type { CheerioAPI } from 'cheerio'; -import { ConcurrencySystem } from '@crawlee/basic'; import type { ConcurrencySystemOptions } from '@crawlee/basic'; import { ContextPipeline } from '@crawlee/basic'; import type { CrawlingContext } from '@crawlee/basic'; @@ -28,13 +27,6 @@ import { RouterHandler } from '@crawlee/basic'; import { RouterRoutes } from '@crawlee/basic'; import { RouteSchemas } from '@crawlee/basic'; import { RoutesFromSchemas } from '@crawlee/basic'; -import { Transform } from 'node:stream'; - -// @public -export function ByteCounterStream(input: { - logTransferredBytes: (transferredBytes: number) => void; - loggingInterval?: number; -}): Transform; // Not exported by the entry point; reachable only as a referenced type. // @public (undocumented) @@ -107,8 +99,6 @@ export interface DOMParseResult { // @public export class FileDownload extends BasicCrawler { constructor(options?: BasicCrawlerOptions); - // (undocumented) - protected buildContextPipeline(): ContextPipeline; } // @public (undocumented) @@ -128,9 +118,6 @@ export interface FileDownloadCrawlingContext export type FileDownloadErrorHandler> = ErrorHandler & ContextExtension>; -// @public (undocumented) -export type FileDownloadHook = InternalHttpHook>; - // @public (undocumented) export type FileDownloadRequestHandler = RequestHandler>; @@ -142,11 +129,6 @@ export class HttpCrawler = constructor(options?: HttpCrawlerOptions & RequireContextPipeline); // (undocumented) protected buildContextPipeline(): ContextPipeline; - protected createDefaultConcurrencySystem(options: ConcurrencySystemOptions): ConcurrencySystem; - // (undocumented) - protected getNavigationTimeoutMillis(): number; - // (undocumented) - protected isRequestBlocked(crawlingContext: InternalHttpCrawlingContext): Promise; } // @public (undocumented) @@ -155,7 +137,7 @@ export interface HttpCrawlerOptions Awaitable>)[]; + postNavigationHooks?: InternalHttpHook[]; preNavigationHooks?: InternalHttpHook, ContextExtension>[]; saveResponseCookies?: boolean; suggestResponseEncoding?: string; @@ -170,10 +152,6 @@ export type HttpErrorHandler> = ErrorHandler & ContextExtension>; -// @public (undocumented) -export type HttpHook = InternalHttpHook>; - // @public (undocumented) export type HttpRequestHandler = RequestHandler>; @@ -194,13 +172,6 @@ JSONData extends JsonValue = any> extends CrawlingContextWithResponse // @public (undocumented) export type InternalHttpHook = (crawlingContext: Context & ContextExtension) => Awaitable>; -// @public -export function MinimumSpeedStream(input: { - minSpeedKbps: number; - historyLengthMs?: number; - checkProgressInterval?: number; -}): Transform; - export * from "@crawlee/basic"; diff --git a/docs/public-api/crawlee-impit-client.api.md b/docs/public-api/crawlee-impit-client.api.md index 67fe777b19ca..36368cd92e74 100644 --- a/docs/public-api/crawlee-impit-client.api.md +++ b/docs/public-api/crawlee-impit-client.api.md @@ -22,7 +22,7 @@ export class ImpitHttpClient extends BaseHttpClient { logger?: CrawleeLogger; }); // (undocumented) - fetch(request: Request, options?: RequestInit & CustomFetchOptions): Promise; + protected fetch(request: Request, options?: RequestInit & CustomFetchOptions): Promise; } // (No @packageDocumentation comment for this package) diff --git a/docs/public-api/crawlee-playwright.api.md b/docs/public-api/crawlee-playwright.api.md index 7496179215d6..f3398de9c970 100644 --- a/docs/public-api/crawlee-playwright.api.md +++ b/docs/public-api/crawlee-playwright.api.md @@ -22,11 +22,8 @@ import type { BrowserType } from 'playwright'; import { Cheerio } from 'cheerio'; import { CheerioAPI } from 'cheerio'; import { Configuration } from '@crawlee/browser'; -import type { ContextPipeline } from '@crawlee/browser'; -import type { ContextPipeline as ContextPipeline_2 } from '@crawlee/basic'; -import type { CrawlingContext } from '@crawlee/browser'; -import type { CrawlingContext as CrawlingContext_2 } from '@crawlee/basic'; -import { Dictionary } from '@crawlee/types'; +import type { CrawlingContext } from '@crawlee/basic'; +import type { Dictionary } from '@crawlee/types'; import type { Download } from 'playwright'; import type { EnqueueLinksOptions } from '@crawlee/basic'; import type { GetUserDataFromRequest } from '@crawlee/browser'; @@ -35,13 +32,12 @@ import { IRequestManager } from '@crawlee/browser'; import type { LaunchOptions } from 'playwright'; import type { LoadedRequest } from '@crawlee/browser'; import type { Page } from 'playwright'; -import { PlaywrightPlugin } from '@crawlee/browser-pool'; +import type { PlaywrightPlugin } from '@crawlee/browser-pool'; import type { RecoverableStatePersistenceOptions } from '@crawlee/core'; import type { RemoteBrowserPool } from '@crawlee/browser-pool'; import type { RemoteBrowserPoolOptions } from '@crawlee/browser-pool'; import type { Request as Request_2 } from '@crawlee/core'; import { Request as Request_3 } from '@crawlee/browser'; -import type { RequestHandler } from '@crawlee/browser'; import type { RequestTransform } from '@crawlee/browser'; import type { Response as Response_2 } from 'playwright'; import type { RouterHandler } from '@crawlee/browser'; @@ -73,8 +69,6 @@ interface AdaptiveHookContext extends Pick, ExtendedContext extends AdaptivePlaywrightCrawlerContext = AdaptivePlaywrightCrawlerContext & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends AdaptivePlaywrightCrawlerStatisticState = AdaptivePlaywrightCrawlerStatisticState> extends BasicCrawler { constructor(options?: AdaptivePlaywrightCrawlerOptions); // (undocumented) - protected buildContextPipeline(): ContextPipeline_2; - // (undocumented) destroy(): Promise; drainRenderingDetections(input?: { timeoutMillis?: number; @@ -83,12 +77,12 @@ export class AdaptivePlaywrightCrawler, Ext // (undocumented) protected init(): Promise; // (undocumented) - protected runRequestHandler(crawlingContext: CrawlingContext_2): Promise; + protected runRequestHandler(crawlingContext: CrawlingContext): Promise; teardown(): Promise; } // @public (undocumented) -export interface AdaptivePlaywrightCrawlerContext extends CrawlingContext_2 { +export interface AdaptivePlaywrightCrawlerContext extends CrawlingContext { // (undocumented) enqueueLinks(options?: EnqueueLinksOptions): Promise; page: Page; @@ -210,6 +204,9 @@ export function fullResultComparator(resultA: StorageTransactionView, resultB: S // @public function gotoExtended(page: Page, request: Request_3, gotoOptions?: PlaywrightDirectNavigationOptions): Promise; +// @public +function handleCloudflareChallenge(page: Page, url: string, options?: HandleCloudflareChallengeOptions): Promise; + // @public export function handleCloudflareChallengeHook(options?: HandleCloudflareChallengeOptions): PlaywrightHook; @@ -311,11 +308,9 @@ interface PlaywrightContextUtils { } // @public -export class PlaywrightCrawler, ExtendedContext extends PlaywrightCrawlingContext = PlaywrightCrawlingContext & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends object = {}> extends BrowserCrawler { +export class PlaywrightCrawler, ExtendedContext extends PlaywrightCrawlingContext = PlaywrightCrawlingContext & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends object = {}> extends BrowserCrawler { constructor(options?: PlaywrightCrawlerOptions); // (undocumented) - protected buildContextPipeline(): ContextPipeline; - // (undocumented) protected navigationHandler(crawlingContext: PlaywrightCrawlingContext, gotoOptions: PlaywrightDirectNavigationOptions): Promise; } @@ -325,7 +320,6 @@ export interface PlaywrightCrawlerOptions, launchContext?: PlaywrightLaunchContext; postNavigationHooks?: BrowserHook>, ContextExtension>[]; preNavigationHooks?: BrowserHook>, ContextExtension>[]; - requestHandler?: RouterHandler | RequestHandler; } // @public (undocumented) @@ -365,6 +359,7 @@ declare namespace playwrightUtils { infiniteScroll, saveSnapshot, parseWithCheerio, + handleCloudflareChallenge, InjectFileOptions, BlockRequestsOptions, PlaywrightDirectNavigationOptions as DirectNavigationOptions, @@ -374,8 +369,7 @@ declare namespace playwrightUtils { SaveSnapshotOptions, HandleCloudflareChallengeOptions, PlaywrightContextUtils, - enqueueLinksByClickingElements, - playwrightUtils_2 as playwrightUtils + enqueueLinksByClickingElements } } diff --git a/docs/public-api/crawlee-puppeteer.api.md b/docs/public-api/crawlee-puppeteer.api.md index 725adc754798..3f368c9c1dc2 100644 --- a/docs/public-api/crawlee-puppeteer.api.md +++ b/docs/public-api/crawlee-puppeteer.api.md @@ -24,20 +24,17 @@ import type { GetUserDataFromRequest } from '@crawlee/browser'; import type { HTTPRequest } from 'puppeteer'; import type { HTTPResponse } from 'puppeteer'; import { IRequestManager } from '@crawlee/browser'; -import type { LaunchOptions } from 'puppeteer'; import type { Page } from 'puppeteer'; import { PuppeteerPlugin } from '@crawlee/browser-pool'; import type { RemoteBrowserPool } from '@crawlee/browser-pool'; import type { RemoteBrowserPoolOptions } from '@crawlee/browser-pool'; import { Request as Request_2 } from '@crawlee/browser'; import type { RequestTransform } from '@crawlee/browser'; -import type { ResponseForRequest } from 'puppeteer'; import type { RouterHandler } from '@crawlee/browser'; import type { RouterRoutes } from '@crawlee/browser'; import type { RouteSchemas } from '@crawlee/browser'; import type { RoutesFromSchemas } from '@crawlee/browser'; import type { SkippedRequestCallback } from '@crawlee/browser'; -import type { Target } from 'puppeteer'; import type { UrlPatternInput } from '@crawlee/browser'; // @public @@ -47,22 +44,16 @@ function addInterceptRequestHandler(page: Page, handler: InterceptHandler): Prom function blockRequests(page: Page, options?: BlockRequestsOptions): Promise; // @public (undocumented) -export interface BlockRequestsOptions { +interface BlockRequestsOptions { extraUrlPatterns?: string[]; urlPatterns?: string[]; } -// @public @deprecated -const blockResources: (page: Page, resourceTypes?: string[]) => Promise; - -// @public @deprecated -function cacheResponses(page: Page, cache: Dictionary>, responseUrlRules: (string | RegExp)[]): Promise; - // @public (undocumented) -export type CompiledScriptFunction = (params: CompiledScriptParams) => Promise; +type CompiledScriptFunction = (params: CompiledScriptParams) => Promise; // @public (undocumented) -export interface CompiledScriptParams { +interface CompiledScriptParams { // (undocumented) page: Page; // (undocumented) @@ -109,7 +100,7 @@ function gotoExtended(page: Page, request: Request_2, gotoOptions?: PuppeteerDir function infiniteScroll(page: Page, options?: InfiniteScrollOptions): Promise; // @public (undocumented) -export interface InfiniteScrollOptions { +interface InfiniteScrollOptions { buttonSelector?: string; maxScrollHeight?: number; scrollDownAndUp?: boolean; @@ -122,7 +113,7 @@ export interface InfiniteScrollOptions { function injectFile(page: Page, filePath: string, options?: InjectFileOptions): Promise; // @public (undocumented) -export interface InjectFileOptions { +interface InjectFileOptions { surviveNavigations?: boolean; } @@ -134,9 +125,6 @@ function injectJQuery(page: Page, options?: { // @public (undocumented) export type InterceptHandler = (request: HTTPRequest) => unknown; -// @public -function isTargetRelevant(page: Page, target: Target): boolean; - // @public export function launchPuppeteer(launchContext?: PuppeteerLaunchContext, configuration?: Configuration): Promise; @@ -158,16 +146,6 @@ export interface PuppeteerBrowserPoolOptions extends Omit; @@ -184,7 +162,7 @@ interface PuppeteerContextUtils { } // @public -export class PuppeteerCrawler, ExtendedContext extends PuppeteerCrawlingContext = PuppeteerCrawlingContext & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends object = {}> extends BrowserCrawler { +export class PuppeteerCrawler, ExtendedContext extends PuppeteerCrawlingContext = PuppeteerCrawlingContext & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends object = {}> extends BrowserCrawler { constructor(options?: PuppeteerCrawlerOptions); // (undocumented) protected buildContextPipeline(): ContextPipeline; @@ -226,22 +204,12 @@ export interface PuppeteerLaunchContext extends BrowserLaunchContext; // @public (undocumented) -export interface SaveSnapshotOptions { +interface SaveSnapshotOptions { configuration?: Configuration; key?: string; keyValueStoreName?: string | null; diff --git a/docs/public-api/crawlee-stagehand.api.md b/docs/public-api/crawlee-stagehand.api.md index a5f66c106753..618da744e204 100644 --- a/docs/public-api/crawlee-stagehand.api.md +++ b/docs/public-api/crawlee-stagehand.api.md @@ -8,7 +8,6 @@ import { Action } from '@browserbasehq/stagehand'; import { ActOptions } from '@browserbasehq/stagehand'; import { ActResult } from '@browserbasehq/stagehand'; import { AgentConfig } from '@browserbasehq/stagehand'; -import { AgentResult } from '@browserbasehq/stagehand'; import type { Browser } from 'playwright'; import type { BrowserController } from '@crawlee/browser-pool'; import { BrowserCrawler } from '@crawlee/browser'; @@ -31,7 +30,6 @@ import type { GetUserDataFromRequest } from '@crawlee/browser'; import type { LaunchContext } from '@crawlee/browser-pool'; import type { LaunchOptions } from 'playwright'; import type { LLMClient } from '@browserbasehq/stagehand'; -import type { LoadedContext } from '@crawlee/browser'; import { ModelConfiguration } from '@browserbasehq/stagehand'; import type { NonStreamingAgentInstance } from '@browserbasehq/stagehand'; import { ObserveOptions } from '@browserbasehq/stagehand'; @@ -56,8 +54,6 @@ export { ActResult } export { AgentConfig } -export { AgentResult } - // @public export function createStagehandRouter = Record>>(routes?: RouterRoutes): RouterHandler; @@ -99,7 +95,7 @@ export interface StagehandBrowserPoolOptions extends Omit, ExtendedContext extends StagehandCrawlingContext = StagehandCrawlingContext & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends object = {}> extends BrowserCrawler { +export class StagehandCrawler, ExtendedContext extends StagehandCrawlingContext = StagehandCrawlingContext & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends object = {}> extends BrowserCrawler { constructor(options?: StagehandCrawlerOptions); // (undocumented) protected buildContextPipeline(): ContextPipeline; @@ -130,13 +126,9 @@ export type StagehandHook = BrowserHook { - launcher?: BrowserType; launchOptions?: LaunchOptions & Parameters[1]; proxyUrl?: string; - stagehandOptions?: StagehandOptions; useChrome?: boolean; - useIncognitoPages?: boolean; - userDataDir?: string; } // @public @@ -174,11 +166,8 @@ class StagehandPlugin extends BrowserPlugin constructor(library: BrowserType, options?: StagehandPluginOptions); protected addProxyToLaunchOptions(launchContext: LaunchContext): Promise; createController(): BrowserController; - getStagehandForBrowser(browser: Browser): Stagehand | undefined; protected isChromiumBasedBrowser(): boolean; protected _launch(launchContext: LaunchContext): Promise; - // (undocumented) - readonly stagehandOptions: StagehandOptions; } // Not exported by the entry point; reachable only as a referenced type. @@ -187,16 +176,6 @@ interface StagehandPluginOptions extends BrowserPluginOptions { stagehandOptions?: StagehandOptions; } -// @public -export interface StagehandRequestHandler extends RequestHandler> { -} - -declare namespace stagehandUtils { - export { - enhancePageWithStagehand - } -} - export * from "@crawlee/browser"; diff --git a/docs/public-api/crawlee-types.api.md b/docs/public-api/crawlee-types.api.md index 8b1b7ad43f75..08a1a3a0ec3a 100644 --- a/docs/public-api/crawlee-types.api.md +++ b/docs/public-api/crawlee-types.api.md @@ -22,14 +22,6 @@ export interface BatchAddRequestsResult { unprocessedRequests: UnprocessedRequest[]; } -// @public (undocumented) -export interface BrowserLikeResponse { - // (undocumented) - headers(): Dictionary; - // (undocumented) - url(): string; -} - // @public (undocumented) export interface Cookie { domain?: string; @@ -164,33 +156,17 @@ export interface HttpRequest { // (undocumented) followRedirect?: boolean | ((response: any) => boolean); // (undocumented) - headerGenerator?: { - getHeaders: (options: Record) => Record; - }; - // (undocumented) - headerGeneratorOptions?: Record; - // (undocumented) headers?: Headers; // (undocumented) - insecureHTTPParser?: boolean; - // (undocumented) - maxRedirects?: number; - // (undocumented) method?: AllowedHttpMethods; // (undocumented) proxyUrl?: string; // (undocumented) - sessionToken?: object; - // (undocumented) signal?: AbortSignal; // (undocumented) - throwHttpErrors?: boolean; - // (undocumented) timeout?: number; // (undocumented) url: string | URL; - // (undocumented) - useHeaderGenerator?: boolean; } // @public @@ -364,12 +340,6 @@ export interface QueueOperationInfo { wasAlreadyPresent: boolean; } -// @public -export type RedirectHandler = (redirectResponse: Response, updatedRequest: { - url?: string | URL; - headers: Headers; -}) => void; - // @public export interface RequestQueueBackend { addBatchOfRequests(requests: RequestSchema[], options?: RequestQueueOperationOptions): Promise; @@ -575,12 +545,6 @@ export type StorageIdentifier = { alias?: never; }; -// @public (undocumented) -export interface StreamOptions extends SendRequestOptions { - // (undocumented) - onRedirect?: RedirectHandler; -} - // @public (undocumented) export interface UnprocessedRequest { // (undocumented) diff --git a/docs/public-api/crawlee-utils.api.md b/docs/public-api/crawlee-utils.api.md index 08a67bf87377..136d0802811b 100644 --- a/docs/public-api/crawlee-utils.api.md +++ b/docs/public-api/crawlee-utils.api.md @@ -106,15 +106,6 @@ export interface MicrodataItem { // @public export type MicrodataValue = string | MicrodataItem; -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -interface NestedSitemap { - // (undocumented) - loc: string; - // (undocumented) - originSitemapUrl: null; -} - // @public (undocumented) export interface OpenGraphProperty { // (undocumented) @@ -138,9 +129,6 @@ export function parseOpenGraph(raw: string, additionalProperties?: OpenGraphProp // @public (undocumented) export function parseOpenGraph($: CheerioAPI, additionalProperties?: OpenGraphProperty[]): Promise>; -// @public (undocumented) -export function parseSitemap(initialSources: SitemapSource[], proxyUrl?: string, options?: T): AsyncIterable; - // @public (undocumented) export interface ParseSitemapOptions { emitNestedSitemaps?: true | false; @@ -198,36 +186,6 @@ export class Sitemap { readonly urls: string[]; } -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -type SitemapSource = ({ - type: 'url'; - url: string; -} | { - type: 'raw'; - content: string; -}) & { - depth?: number; -}; - -// @public (undocumented) -export type SitemapUrl = SitemapUrlData & { - originSitemapUrl: string; -}; - -// Not exported by the entry point; reachable only as a referenced type. -// @public (undocumented) -interface SitemapUrlData { - // (undocumented) - changefreq?: 'always' | 'hourly' | 'daily' | 'weekly' | 'monthly' | 'yearly' | 'never'; - // (undocumented) - lastmod?: Date; - // (undocumented) - loc: string; - // (undocumented) - priority?: number; -} - // @public export function sleep(millis?: number): Promise; diff --git a/docs/public-api/crawlee.api.md b/docs/public-api/crawlee.api.md index e6b955a2952f..bf251450a59c 100644 --- a/docs/public-api/crawlee.api.md +++ b/docs/public-api/crawlee.api.md @@ -4,31 +4,9 @@ ```ts -import { downloadListOfUrls } from '@crawlee/utils'; -import { extractMicrodata } from '@crawlee/utils'; -import { Log } from '@crawlee/core'; -import { parseOpenGraph } from '@crawlee/utils'; -import { playwrightUtils } from '@crawlee/playwright'; -import { puppeteerUtils } from '@crawlee/puppeteer'; -import { sleep } from '@crawlee/utils'; -import { social } from '@crawlee/utils'; - -// @public (undocumented) -export const utils: { - puppeteer: typeof puppeteerUtils; - playwright: typeof playwrightUtils; - log: Log; - social: typeof social; - sleep: typeof sleep; - downloadListOfUrls: typeof downloadListOfUrls; - parseOpenGraph: typeof parseOpenGraph; - extractMicrodata: typeof extractMicrodata; -}; - export * from "@crawlee/basic"; export * from "@crawlee/browser"; -export * from "@crawlee/browser-pool"; export * from "@crawlee/cheerio"; export * from "@crawlee/core"; export * from "@crawlee/fs-storage"; diff --git a/docs/upgrading/upgrading_v4.md b/docs/upgrading/upgrading_v4.md index 6acd734ef627..560adbf2509f 100644 --- a/docs/upgrading/upgrading_v4.md +++ b/docs/upgrading/upgrading_v4.md @@ -178,7 +178,7 @@ const crawler = new PlaywrightCrawler({ }); ``` -The browser *launchers* (`PlaywrightLauncher`, `PuppeteerLauncher`) keep their `(launchContext?, configuration?)` constructor signature — this change is only about the crawler classes. +`PuppeteerLauncher` keeps its `(launchContext?, configuration?)` constructor signature — this change is only about the crawler classes. `PlaywrightLauncher` is no longer exported at all, see [`PlaywrightLauncher` is no longer exported](#playwrightlauncher-is-no-longer-exported). ### Configuration class redesign @@ -226,8 +226,8 @@ The following methods and properties have been removed from `Configuration`: - `Configuration.getEventManager()` - moved to `ServiceLocator.getEventManager()` - `Configuration.useStorageClient()` - use `ServiceLocator.setStorageBackend()` instead - `Configuration.useEventManager()` - use `ServiceLocator.setEventManager()` instead -- `Configuration.resetGlobalState()` - use `serviceLocator.reset()` instead -- `Configuration.storageManagers` - moved to `ServiceLocator.getStorageInstanceManager()` +- `Configuration.resetGlobalState()` - use `serviceLocator.reset()` instead. The method is marked `@internal`: it exists at runtime and is the supported way to tear down global state between tests, but it is not part of the public API and may change without a major version bump. +- `Configuration.storageManagers` - use `serviceLocator.getStorageInstanceManager()` instead. Also marked `@internal` - available at runtime, but excluded from the documented surface and not covered by semver guarantees. Application code should reach storages through `Dataset.open()` / `KeyValueStore.open()` / `RequestQueue.open()` rather than the instance manager. The `EventManager` and `LocalEventManager` constructors now accept an options object for configuring event intervals (e.g. `persistStateIntervalMillis`, `systemInfoIntervalMillis`). You can also use the new `LocalEventManager.fromConfiguration()` factory method to create an instance with intervals derived from a `Configuration` object. @@ -335,7 +335,7 @@ const store = await KeyValueStore.open(null, { config: new Configuration({ persi const store = await KeyValueStore.open(null, { configuration: new Configuration({ persistStorage: false }) }); ``` -Renamed properties — `Dataset.config`, `KeyValueStore.config`, `Snapshotter.config` and `BrowserLauncher.config` (including `PlaywrightLauncher`, `PuppeteerLauncher` and `StagehandLauncher`) are now `.configuration`. +Renamed properties — `Dataset.config`, `KeyValueStore.config`, `Snapshotter.config` and `BrowserLauncher.config` (including `PuppeteerLauncher`) are now `.configuration`. The `configuration` crawler option is unchanged, as are `serviceLocator.getConfiguration()` and `serviceLocator.setConfiguration()`. @@ -911,8 +911,6 @@ The leading underscore was dropped from protected and private class members acro - `BasicCrawler._runRequestHandler` -> `BasicCrawler.runRequestHandler` -The file-system storage backends' shared `CachedIdClient._cachedId` protected field was also renamed to `cachedId` (this only affects custom `@crawlee/fs-storage` backends that subclass it). - If you subclass a crawler or implement a custom browser plugin, these `protected` extension points lost their underscore too: - `BasicCrawler._init` -> `init` @@ -947,7 +945,7 @@ This is intentional: these were never a supported API. If you relied on overridi The change spans, among others: -- **`BasicCrawler`** — `unexpectedStop`, `requestHandlerTimeoutMillis`, `sameDomainDelayMillis`, `domainAccessedTime`, `handledRequestsCount`, `statusMessageLoggingInterval`, `statusMessageCallback`, `ignoreHttpErrorStatusCodes`, `taskLoopOptions` (was `autoscaledPoolOptions`), `autoscaledPool`, `respectRobotsTxtFile`, and the helpers `buildBasicContextPipeline`, `validateRequestUserData`, `pauseOnMigration`, `fetchNextRequest`, `delayRequest`, `handleRequest`, `timeoutAndRetry`, `isTaskReadyFunction`, `defaultIsFinishedFunction`, `requestFunctionErrorHandler`, `handleFailedRequestHandler`, `canRequestBeRetried` +- **`BasicCrawler`** — `running`, `hasFinishedBefore`, `basicContextPipeline`, `unexpectedStop`, `requestHandlerTimeoutMillis`, `sameDomainDelayMillis`, `domainAccessedTime`, `handledRequestsCount`, `statusMessageLoggingInterval`, `statusMessageCallback`, `ignoreHttpErrorStatusCodes`, `taskLoopOptions` (was `autoscaledPoolOptions`), `autoscaledPool`, `respectRobotsTxtFile`, and the helpers `buildBasicContextPipeline`, `validateRequestUserData`, `pauseOnMigration`, `fetchNextRequest`, `delayRequest`, `handleRequest`, `timeoutAndRetry`, `isTaskReadyFunction`, `defaultIsFinishedFunction`, `requestFunctionErrorHandler`, `handleFailedRequestHandler`, `canRequestBeRetried` - **`HttpCrawler`** — `preNavigationHooks`, `postNavigationHooks`, `saveResponseCookies`, `navigationTimeoutMillis`, `suggestResponseEncoding`, `forceResponseEncoding`, `supportedMimeTypes`, and the helpers `requestFunction`, `parseResponse`, `getRequestOptions`, `encodeResponse`, `extendSupportedMimeTypes`, `handleRequestTimeout` - **`AutoscaledPool`** — the whole class is `@internal` in v4, so its members are not enumerated here; see [`AutoscaledPool` is no longer public API](#autoscaledpool-is-no-longer-public-api) - **`SessionPool`** — all pool internals (`log`, `maxPoolSize`, `createSessionFunction`, `keyValueStore`, `sessions`, `sessionMap`, `sessionOptions`, `persistStateKey`, `persistStateKeyValueStoreId`, `events`, `persistenceOptions`, `sessionReuseStrategy`, and the helpers `ensureInitialized`, `maybeLoadSessionPool`, `registerSession`, `createSession`, `hasSpaceForSession`, `pickSession`, `removeRetiredSessions`, `getRandomIndex`, `defaultCreateSessionFunction`) @@ -965,7 +963,25 @@ The change spans, among others: - **`BrowserCrawler`** — `navigationTimeoutMillis`, `preNavigationHooks`, `postNavigationHooks`, `saveResponseCookies` (now `private readonly`; configure them through the constructor options as before), and the helpers `isRequestBlocked`, `applyCookies` (was `_applyCookies`), `handleNavigationTimeout` (was `_handleNavigationTimeout`), `throwIfProxyError` (was `_throwIfProxyError`) - **`BrowserLauncher`** — the helpers `getChromeExecutablePath`, `getTypicalChromeExecutablePath`, `validateProxyUrlProtocol` (were `_`-prefixed). `getDefaultHeadlessOption` (was `_getDefaultHeadlessOption`) stays `protected` — it is an override point (`PuppeteerLauncher` overrides it) — but lost its underscore prefix - **`RobotsTxtFile.load` and `Sitemap.parse`** — internal static factory helpers, now `private` (use the public `RobotsTxtFile.from` / `Sitemap.load` / `Sitemap.fromXmlString` entry points) -- Various internal fields on `BrowserController` (`id`, `browserPlugin`, `log`) and `BrowserPlugin` (`name`, `library`, `launchOptions`, `proxyUrl`, `userDataDir`, `browserPerProxy`, `ignoreProxyCertificate`, `log`) are now `readonly` +- Various internal fields on `BrowserController` (`id`, `browserPlugin`) and `BrowserPlugin` (`name`, `library`, `launchOptions`, `proxyUrl`, `userDataDir`, `browserPerProxy`, `ignoreProxyCertificate`) are now `readonly` (the `log` field on both is off the public surface entirely — see [`BrowserPool` internals are private](#browserpool-internals-are-private)) + +The `running` and `hasFinishedBefore` flags on `BasicCrawler` were internal run-state bookkeeping for the re-run logic. If you were polling `crawler.running` to tell whether a crawl was in progress, track that yourself around the `crawler.run()` promise instead. + +### Declarations that are now `@internal` + +These are still present at runtime, but they are excluded from the documented surface and can change without a major version bump: `Session.getState()`, `SessionOptions.log`, `Router.getTimeoutSecs()`, `Router.getMaxTimeoutSecs()`, `Request.skippedReason`, `RestrictedCrawlingContext.id`, `ThrottlingRequestManager.drop()`, `ThrottlingRequestManager.innerManager`, `ContextPipelineInitializationError`, `ContextPipelineCleanupError` and `RequestHandlerError`. + +`BLOCKED_STATUS_CODES` is now typed `readonly number[]`. Copy it (`[...BLOCKED_STATUS_CODES]`) if you were mutating it. + +### `HttpCrawler`, `FileDownload` and `JSDOMCrawler` internals are no longer accessible to subclasses + +`HttpCrawler.isRequestBlocked` and `FileDownload.buildContextPipeline` are now `private`. Block detection is configured through `retryOnBlocked` and `blockedStatusCodes`; to add your own checks, throw a `SessionError` from a `postNavigationHook`. To extend the `FileDownload` context pipeline, pass a `contextPipelineBuilder` to the constructor rather than subclassing. + +`HttpCrawler.buildContextPipeline`, `HttpCrawler.getNavigationTimeoutMillis`, `HttpCrawler.createDefaultConcurrencySystem` and `JSDOMCrawler.buildContextPipeline` stay `protected` and overridable, but are now marked `@internal` — they are implementation seams shared with `@crawlee/cheerio`, `@crawlee/jsdom` and `@crawlee/linkedom`, and their signatures may change in a minor release. + +### The protected `getMessageFromError()` returns `string` + +`BasicCrawler.getMessageFromError()` previously widened its return type to `string | TimeoutError | undefined`. Overrides must now return a `string`, and callers can drop any `as string` casts. ### The `RequestQueue` constructor no longer takes a `Configuration` @@ -981,6 +997,21 @@ The exported handler types were reshaped accordingly. `ErrorHandler` and `Reques `PlaywrightHook`, `PuppeteerHook` and `StagehandHook` are now type aliases (previously interfaces) generic over the request's `userData` type. A hook that types its context — via the generic (e.g. `PlaywrightHook`) or an explicit context annotation — is now assignable to the `preNavigationHooks` / `postNavigationHooks` options of an untyped crawler. If you extended one of these interfaces, use an intersection type instead. +### Removed navigation hook type aliases + +The `HttpHook`, `CheerioHook`, `JSDOMHook`, `LinkeDOMHook` and `FileDownloadHook` type aliases were removed. None of them could actually type a `preNavigationHooks` or `postNavigationHooks` entry: both options are typed over the *pre-navigation* crawling context, while these aliases closed over the fully parsed one — so a function annotated with them required the pre-navigation context to already supply `$`, `body` or `window`, and assignment was rejected. (`FileDownload` has no hook options at all.) + +Let the hook be inferred from the options object, or annotate its parameter with the crawler's own context type: + +```diff +-const hook: CheerioHook = async (ctx) => { /* ... */ }; ++const hook = async (ctx: CheerioCrawlingContext) => { /* ... */ }; + + new CheerioCrawler({ preNavigationHooks: [hook] }); +``` + +To type a standalone hook against the pre-navigation context, use `InternalHttpHook`. + ### `RecoverableState` reshaped `serialize` and `deserialize` now take and return values rather than strings (so v3 records will not load) and each also accept a [Standard Schema](https://standardschema.dev) — a zod codec works as `deserialize` directly — `reset()` is synchronous, no longer clears the persisted record (the new `resetStore()` does) and doubles as a way to establish the state without awaiting `initialize()`, `persistStateKvsName` and `persistStateKvsId` collapsed into a single `keyValueStore` option taking a store (or a pending `KeyValueStore.open()`), `defaultState` also accepts a factory (which you need for a state `structuredClone` cannot rebuild, as the deep copy no longer goes through `serialize`/`deserialize`), and there is a new `persistenceTimeoutMillis` option. `teardown()` is no longer terminal — `initialize()` can be called again to open another persistence window — and a write that fails during a periodic `PERSIST_STATE` or during `teardown()` is warned about rather than thrown. A direct `persistState()` still throws. @@ -1212,6 +1243,17 @@ await sharedPool.destroy(); The `crawler.browserPool` property is now **read-only** (a getter). It was previously a writable field, so any code that reassigned it after construction (`crawler.browserPool = myPool`) no longer works — pass your pool via the `browserPool` constructor option instead. +### `@crawlee/browser-pool` is no longer re-exported from `crawlee` + +The meta-package used to `export *` from `@crawlee/browser-pool`, which made `BrowserPool`, `PuppeteerPlugin`, `PlaywrightPlugin`, `BrowserController`, `LaunchContext`, the fingerprint enums and the `IBrowserPool` / `NewPageOptions` interfaces importable from `crawlee`. They no longer are. Add `@crawlee/browser-pool` to your dependencies and import them from there: + +```diff +-import { BrowserPool, PlaywrightPlugin } from 'crawlee'; ++import { BrowserPool, PlaywrightPlugin } from '@crawlee/browser-pool'; +``` + +The crawler-facing surface is unaffected: `playwrightBrowserPool()`, `puppeteerBrowserPool()` and their `remote*` counterparts still come from `crawlee` (and from `@crawlee/playwright` / `@crawlee/puppeteer`), so the common case of building a pool for a crawler needs no extra dependency. + ### `BrowserCrawlingContext.browserController` has been removed The `browserController` property is no longer part of the crawling context (`BrowserCrawlingContext`). Browser controller management is now fully internal to the pool — the crawler interacts with the pool only through the `IBrowserPool` interface (`newPage`, `closePage`, `extractPageState`, and `injectPageState`). @@ -1325,6 +1367,55 @@ postNavigationHooks: [ If you called the standalone `playwrightUtils.handleCloudflareChallenge(page, url, session, options)` directly, note that the `session` parameter is gone - the v4 signature is `handleCloudflareChallenge(page, url, options)`, so an options object passed in the old fourth position would be silently ignored. +The `playwrightUtils` namespace also used to contain a nested, self-referential `playwrightUtils` object; it is gone. `handleCloudflareChallenge` was previously only reachable as `playwrightUtils.playwrightUtils.handleCloudflareChallenge` — `playwrightUtils.handleCloudflareChallenge(page, url, options)` now works as documented. + +### `PlaywrightLauncher` is no longer exported + +`PlaywrightLauncher` was an implementation detail of `launchPlaywright()` and the Playwright browser pools, and it is no longer part of `@crawlee/playwright`'s (or `crawlee`'s) public exports. Use `launchPlaywright(launchContext, configuration)` to get a `Browser`, or `playwrightBrowserPool()` / `remotePlaywrightBrowserPool()` when you need a pool. `PlaywrightLaunchContext` is still exported, so the options object can still be typed. + +### `BrowserCrawler.launchContext` was removed + +The `launchContext` property on crawler instances is gone. It was assigned once in the constructor and never read, so nothing in Crawlee consumed it. The `launchContext` **option** is unchanged on `PlaywrightCrawler`, `PuppeteerCrawler` and `StagehandCrawler` — only the echo of it on the instance is gone. If you were reading `crawler.launchContext`, keep your own reference to the object you passed in. + +The abstract `BrowserCrawler` base class also lost its third type parameter, `LaunchOptions`, which existed only to type that field. Custom crawlers extending `BrowserCrawler` must drop that type argument: + +```diff +-class MyCrawler extends BrowserCrawler { ++class MyCrawler extends BrowserCrawler { +``` + +### Unused browser options are now rejected instead of ignored + +`PlaywrightCrawler` no longer accepts a top-level `launcher` option, and `PlaywrightLaunchContext` no longer accepts `launchContextOptions`; both were silently ignored and are now reported as unknown options by the constructors' validation. Pass the browser type as `launchContext.launcher`, and persistent-context settings inside `launchContext.launchOptions`. + +### `LauncherBrowserPoolOptions` and `LauncherRemoteBrowserPoolOptions` are no longer exported + +These two type aliases described the options accepted by `BrowserLauncher.createBrowserPool()` and `.createRemoteBrowserPool()`, both internal APIs. Use the per-library option types instead — `PlaywrightBrowserPoolOptions` / `RemotePlaywrightBrowserPoolOptions`, `PuppeteerBrowserPoolOptions` / `RemotePuppeteerBrowserPoolOptions`, `StagehandBrowserPoolOptions` / `RemoteStagehandBrowserPoolOptions` — which are the caller-facing types for `playwrightBrowserPool()`, `puppeteerBrowserPool()` and `stagehandBrowserPool()`. + +### Dead v3 fingerprinting types are removed + +`BrowserSpecification`, `GetFingerprintReturn` and `@crawlee/browser-pool`'s own `FingerprintGenerator` interface (which shadowed `fingerprint-generator`'s class of the same name) are no longer exported. They had no consumers — `BrowserPool.fingerprintGenerator` is typed by `fingerprint-generator`'s `FingerprintGenerator`, and `fingerprintGeneratorOptions` is still typed by `FingerprintGeneratorOptions`. + +### `BROWSER_POOL_EVENTS.BROWSER_CLOSED` was removed + +It was never emitted. Listen for `BROWSER_CONTROLLER_EVENTS.BROWSER_CLOSED` on a `BrowserController` instead. + +### `BrowserPool` internals are private + +`pages`, `pageIds`, `pageCounter`, `pageToBrowserController`, `startingBrowserControllers`, `activeBrowserControllers`, `retiredBrowserControllers`, `operationTimeoutMillis`, `closeInactiveBrowserAfterMillis`, `maxOpenPagesPerBrowser`, `retireBrowserAfterPageCount` and `useFingerprints` are no longer readable or writable from outside the pool. Use `getPage()`, `getPageId()` and `getBrowserControllerByPage()` to reach pages, the `browserLaunched` / `browserRetired` / `pageCreated` / `pageClosed` events to observe the pool's lifecycle, and pass the corresponding `BrowserPoolOptions` at construction instead of assigning to the mirrors afterwards — the options themselves are unchanged. + +The six hook arrays (`preLaunchHooks`, `postLaunchHooks`, `prePageCreateHooks`, `postPageCreateHooks`, `prePageCloseHooks`, `postPageCloseHooks`) are private too. Supply hooks through the constructor options; mutating or replacing the arrays on a live pool is no longer possible. + +```diff +-const browserPool = new BrowserPool({ browserPlugins: [plugin] }); +-browserPool.postLaunchHooks.push(myHook); ++const browserPool = new BrowserPool({ browserPlugins: [plugin], postLaunchHooks: [myHook] }); +``` + +`BrowserController.log` and `BrowserPlugin.log` are no longer part of the public type surface either. Subclasses inside `@crawlee/browser-pool` still use them, but they are not covered by backwards-compatibility guarantees — get your own logger from `serviceLocator.getLogger()`. + +Several more members are `@internal`, and remain present at runtime only: `BrowserPool.fingerprintInjector`, `.fingerprintCache`, `.fingerprintOptions`, `.maxOpenBrowsers`, `.hasFreeBrowserSlot()`, `.hasActiveBrowserWithFreeCapacity()`; `BrowserController.isActive`, `.totalPages`, `.lastPageOpenedAt`, `.normalizeProxyOptions()`; the `PlaywrightPlugin` / `PuppeteerPlugin` `useRemoteConnection()` overrides; `RemoteBrowserPool.browserPool`; `RemoteBrowserPoolOptions.slotPollIntervalMillis`; `LaunchContextOptions.isRemote`; `AnonymizeProxySugarOptions`; and the `PlaywrightBrowser` constructor. The `_close` / `_kill` / `_newPage` / `_getCookies` / `_setCookies` hooks on `BrowserController` and the `_launch` / `addProxyToLaunchOptions` / `isChromiumBasedBrowser` hooks on `BrowserPlugin` are *not* in that list: they carry no release tag, on the abstract declarations and on the `PlaywrightController` / `PuppeteerController` / `PlaywrightPlugin` / `PuppeteerPlugin` overrides alike, and remain the extension contract for your own subclasses. + ## Only if you customize crawler statistics Applies when you passed `statisticsOptions` to a crawler, subclassed `Statistics`, or passed type arguments to `BrowserCrawler`/`BrowserCrawlerOptions`. @@ -1383,7 +1474,7 @@ Applies when you implemented `BaseHttpClient` yourself, or imported `gotScraping The HTTP client abstraction moved out of `@crawlee/core` into two new packages, and its shape changed to match the native `fetch` model. -- **`@crawlee/http-client`** (new) now owns the `BaseHttpClient` abstract base class, along with `FetchHttpClient`, `ResponseWithUrl` / `IResponseWithUrl`, and `CustomFetchOptions`. +- **`@crawlee/http-client`** (new) now owns the `BaseHttpClient` abstract base class, along with `FetchHttpClient`, `ResponseWithUrl`, and `CustomFetchOptions`. - **`@crawlee/got-scraping-client`** (new) provides `GotScrapingHttpClient` — the `got-scraping`-backed client — as an opt-in dependency, so `got-scraping` is no longer pulled into every install. `BaseHttpClient` was redesigned around `fetch`. In v3 it declared `sendRequest(request): Promise` and `stream(request): Promise`; in v4 subclasses implement a single `protected abstract fetch(input: Request, init?): Promise` and the base class provides `sendRequest(request, options?): Promise`. There is no `stream()` method anymore — a `Response` already exposes `body` as a stream. The following symbols that were part of the old `@crawlee/core` HTTP surface are **removed**: `HttpResponse`, `HttpResponseWithoutBody`, `StreamingHttpResponse`, `ResponseTypes`, `BaseHttpResponseData`, `SimpleHeaders`, and `processHttpRequestOptions`. @@ -1411,6 +1502,34 @@ class MyClient extends BaseHttpClient { } ``` +#### `IResponseWithUrl` is removed and `ResponseWithUrl.url` is `readonly` + +The `IResponseWithUrl` interface exported by `@crawlee/http-client` has been removed. It only ever described `Response & { url: string }`, which is exactly what the concrete `ResponseWithUrl` class provides — use `ResponseWithUrl` (or plain `Response`) in its place. + +`ResponseWithUrl.url` is now `readonly`. Pass the URL through the constructor (`new ResponseWithUrl(body, { url, status, headers })`) rather than assigning to it afterwards; nothing in Crawlee ever mutated it. + +#### `fetch` is protected on the built-in HTTP clients + +`BaseHttpClient` has always declared `protected abstract fetch(input, init?)`, but `FetchHttpClient`, `GotScrapingHttpClient` and `ImpitHttpClient` accidentally re-declared it without the modifier, making the raw network call publicly reachable. All three are now `protected override`, matching the base contract. + +If you were calling `httpClient.fetch(request, options)` directly to bypass Crawlee's cookie and redirect handling, use `httpClient.sendRequest(request, options)` instead — it accepts `session`, `cookieJar`, `proxyUrl`, `timeoutMillis`, `signal` and `ignoreTlsErrors` and returns the final `Response`. Custom clients extending `BaseHttpClient` are unaffected: overriding `fetch` as `protected` was already the documented shape. + +#### Removed `@crawlee/types` HTTP types + +The `StreamOptions` and `RedirectHandler` types have been removed. They were the options and redirect-callback types for `BaseHttpClient.stream()`, which no longer exists in v4 — a custom HTTP client now only implements `sendRequest(request: Request, options?: SendRequestOptions)`. If you referenced either type, delete the reference; there is no replacement. + +The `BrowserLikeResponse` interface has been removed. It was a v3-era shim for reading `url()` and `headers()` off a got-style response during cookie handling, and has had no consumer since HTTP responses became standard `Response` objects. Read `response.url` and `response.headers` directly instead. + +#### Removed `HttpRequest` properties + +The following properties have been removed from `HttpRequest`, as v4's HTTP client contract is `fetch`-shaped and read none of them: + +- `headerGenerator`, `headerGeneratorOptions`, `useHeaderGenerator` — header and fingerprint generation is now the HTTP client implementation's concern. `ImpitHttpClient` derives headers from the session fingerprint automatically; configure it through the client's own constructor options. +- `sessionToken` — fingerprint stability is now keyed off the `Session` passed via `SendRequestOptions.session`. +- `insecureHTTPParser` — no longer supported; there is no equivalent in the `fetch`-based clients. +- `throwHttpErrors` — use the crawler's `additionalHttpErrorStatusCodes` / `ignoreHttpErrorStatusCodes` options, or inspect `response.status` yourself. +- `maxRedirects` — redirect following is handled inside `BaseHttpClient.sendRequest` and capped at 10 redirects; it is no longer configurable per request. + #### `gotScraping` is no longer exported from `@crawlee/utils` The `gotScraping` singleton previously exported from `@crawlee/utils` has been removed. If you used it as the crawler's HTTP client, use the new `GotScrapingHttpClient` instead: @@ -1648,6 +1767,11 @@ await enqueueLinks({ urls, requestQueue }); await enqueueLinks({ urls, requestManager }); ``` +#### Removed loader and manager type aliases + +- `RequestListSource`, `UrlList` and `NewUrlOptions` are gone; the signatures that used them now spell their types out inline (`(string | Source)[]`, `(string | null)[]` and `{ request?: Request }` respectively). No behavioral change — replace the alias with the expansion if you referenced it. +- `RequestManagerOpener` is no longer exported, along with the `ThrottlingRequestManagerOptions.requestManagerOpener` option that took one. + ## Only if you configure or implement storage backends The implicit default (file-system storage under `./storage`) behaves as before — this section matters when you construct a storage backend explicitly or implement your own. @@ -1864,6 +1988,21 @@ Because the in-memory queue lives entirely within a single process and is never `MemoryStorageBackend` never accepted `writeMetadata` (it has no on-disk format to begin with), so there is nothing to change there. +#### `FileSystemStorageBackend` exposes no directory fields + +`FileSystemStorageBackend` no longer exposes `localDataDirectory`, `datasetsDirectory`, `keyValueStoresDirectory` or `requestQueuesDirectory` as readable properties — the `StorageBackend` interface declares only methods, and these were never part of it. The on-disk layout is unchanged, so join the paths yourself from the directory you configured: + +```typescript +import { resolve } from 'node:path'; + +const localDataDirectory = './storage'; +const storageBackend = new FileSystemStorageBackend({ localDataDirectory }); + +const datasetsDirectory = resolve(localDataDirectory, 'datasets'); +const keyValueStoresDirectory = resolve(localDataDirectory, 'key_value_stores'); +const requestQueuesDirectory = resolve(localDataDirectory, 'request_queues'); +``` + #### Out-of-band key-value files (e.g. a hand-placed `INPUT.json`) Keys are literal. `aaa` and `aaa.json` are two distinct keys, and `FileSystemStorageBackend` never infers a key from a file's extension — in v3 a hand-placed `aaa.json` was readable as `aaa`, in v4 it is not. @@ -1881,6 +2020,25 @@ Beyond the literal keys, three v3 behaviors are gone: - **Extensionless input files report `application/octet-stream`.** In v3 a bare value file with no extension was read as `text/plain`. Give the file a `.json` extension if you need a more specific type. - **Malformed input files are no longer silently swallowed.** In v3 an `INPUT.json` containing invalid JSON was treated as a missing record (`getValue` returned `undefined`). In v4 the raw bytes are returned verbatim and parsing happens in the `KeyValueStore` frontend, so a malformed value surfaces a parse error at read time instead of looking absent. +### Storage internals are no longer part of the public API + +Several declarations that were only ever implementation details of `Dataset`, `KeyValueStore`, `RequestQueue` and `StorageTransaction` are no longer exported or no longer documented: + +- `StorageStatsTracker` is no longer exported. Read the counters through the `stats` getter on each storage instead — `dataset.stats`, `store.stats`, `queue.stats` — whose types (`DatasetStats`, `KeyValueStoreStats`, `RequestQueueStats`) remain public. +- `resolveStorageIdentifier()` is no longer exported. Use `Dataset.open()` / `KeyValueStore.open()` / `RequestQueue.open()`, which accept the same `id` / `name` / `alias` identifier forms. +- `DatasetOptions`, `KeyValueStoreOptions` and `RequestQueueOptions` are internal. They only described the arguments of the storage constructors, which were already internal — always open storages through the static `open()` methods. +- `StorageTransaction.journal` and `StorageTransaction.policy`, along with the journal entry types (`JournalEntry`, `DatasetJournalEntry`, `KeyValueStoreJournalEntry`, `RequestQueueJournalEntry`, `JournaledRequest`), are internal. For read-only introspection of a transaction use the `StorageTransactionView` accessors: `datasetItems`, `enqueuedUrls`, `keyValueStoreChanges`. + +### `Dataset` field visibility now matches its siblings + +- `Dataset.backend` is private, matching `KeyValueStore.backend`. Use the `Dataset` methods (`pushData`, `getData`, `getInfo`, `drop`, ...) rather than reaching for the backend client. +- `Dataset.id` and `Dataset.name` are `readonly`, matching `KeyValueStore` and `RequestQueue`. +- `Dataset.log` has been removed. It was never read by Crawlee and `KeyValueStore` never had it; use your own logger, or `crawler.log` inside a request handler. + +### `MemoryStorageBackend.createRequestQueueBackend()` returns the `RequestQueueBackend` interface + +It is now typed with the `RequestQueueBackend` interface from `@crawlee/types`, like its two sibling factories and like `FileSystemStorageBackend`. The runtime object is unchanged, but memory-only members (`listItems()`, `cacheKey`, `handledRequestCount`, `pendingRequestCount`, ...) are no longer visible through the return type. Read queue counts from `await getMetadata()`. + ## Only if you tuned autoscaling The `minConcurrency` / `maxConcurrency` / `maxRequestsPerMinute` crawler options work as before. This section matters when you used `autoscaledPoolOptions`, drove an `AutoscaledPool` directly, or configured snapshotting and system status. @@ -1957,6 +2115,8 @@ const crawler = new CheerioCrawler({ }); ``` +`ConcurrencySystem.desiredConcurrency` is a **read-only getter** — the setter is gone. The value is owned by the autoscaler, which recomputes it from the load signals on every tick, so any write was overwritten within one `autoscaleIntervalSecs`. Set the starting point with the `desiredConcurrency` constructor option, and retune a running system through `minConcurrency` / `maxConcurrency`, which both clamp `desiredConcurrency` into the new bounds immediately. + `crawler.pause()` resolves once the requests already in flight have settled, and leaves `run()` pending until you `resume()` — unlike `crawler.stop()`, which ends the run gracefully. One behavioral consequence of the split: pausing no longer suspends autoscaling, because the autoscaling interval belongs to the `ConcurrencySystem`, which knows nothing about its borrowers' pause state — deliberately, since other crawlers sharing it may still need scaling. A paused crawler's system keeps evaluating (and possibly scaling down) the desired concurrency and keeps emitting its periodic state log. Scaling *up* stays effectively blocked, as the current concurrency drains below the ratio required for a scale-up. To silence the system during a long pause, `stop()` it (if you own it) and `start()` it again before resuming; a restart discards the snapshots taken before it, so the pause is not mistaken for load. ##### If you were driving an `AutoscaledPool` directly @@ -2141,11 +2301,12 @@ The general-purpose utility types owned by `@crawlee/types` are no longer re-exp Besides the resource-detection helpers above, several other `@crawlee/utils` exports were removed or moved: - **Removed URL helpers:** `filterUrl(target, origin, strategy)`, `matchesEnqueueStrategy(strategy, target, origin)`, and the `UNSUPPORTED_SCHEME_MESSAGE` constant. URL filtering by enqueue strategy is now internal to `enqueueLinks`. The related `filterRequestsByPatterns(requests, patterns?, onSkippedUrl?)` function (from `@crawlee/core`) was removed for the same reason — pattern-based request filtering now happens inside `enqueueLinks`. -- **Relocated enums/types:** `EnqueueStrategy` now lives in `@crawlee/core` and `SearchParams` in `@crawlee/types`. They are no longer re-exported from `@crawlee/utils`, so `import { EnqueueStrategy } from '@crawlee/utils'` breaks — import them from `crawlee` (the meta-package) or from `@crawlee/core` / `@crawlee/types` instead. +- **Relocated enums/types:** `EnqueueStrategy` is now exported from `@crawlee/core`, `SearchParams` from `@crawlee/types`. They are no longer re-exported from `@crawlee/utils`, so `import { EnqueueStrategy } from '@crawlee/utils'` breaks — import them from `crawlee` (the meta-package) or from `@crawlee/core` / `@crawlee/types` instead. - **Removed `RobotsFile` alias:** `RobotsFile` was an alias for the `RobotsTxtFile` class and is removed. Rename any usage to `RobotsTxtFile`; the class itself is unchanged apart from the signature change described below. -- **Split into public and `/internal` entry points:** the main `@crawlee/utils` entry now exposes only the user-facing helpers (`sleep`, `htmlToText`, `extractUrls`, `downloadListOfUrls`, `expandShadowRoots`, the `social` namespace, the Open Graph parser, and the robots/sitemap utilities). Helpers that primarily serve the crawler packages - e.g. `URL_NO_COMMAS_REGEX`, `URL_WITH_COMMAS_REGEX`, `extractUrlsFromCheerio`, `tryAbsoluteURL`, and the blocked-detection and iterable helpers - moved to the `@crawlee/utils/internal` entry point. They keep working, but imports need updating: `import { URL_NO_COMMAS_REGEX } from '@crawlee/utils/internal'`. As the name suggests, the internal entry follows no semver guarantees. +- **Split into public and `/internal` entry points:** the main `@crawlee/utils` entry now exposes only the user-facing helpers (`sleep`, `htmlToText`, `extractUrls`, `downloadListOfUrls`, the `social` namespace, the Open Graph parser, and the robots/sitemap utilities `RobotsTxtFile`, `Sitemap` and `discoverValidSitemaps`). Helpers that primarily serve the crawler packages - e.g. `URL_NO_COMMAS_REGEX`, `URL_WITH_COMMAS_REGEX`, `extractUrlsFromCheerio`, `tryAbsoluteURL`, `expandShadowRoots`, and the blocked-detection and iterable helpers - moved to the `@crawlee/utils/internal` entry point, which carries no semver guarantees. They keep working, but imports need updating: `import { URL_NO_COMMAS_REGEX } from '@crawlee/utils/internal'`. - **Removed `CheerioRoot` and the cheerio type re-exports:** `CheerioRoot` was an alias for cheerio's own `CheerioAPI` and is gone; `parseWithCheerio()` and `htmlToText()` are typed with `CheerioAPI` directly. The crawler packages also no longer re-export `Cheerio`, `CheerioAPI` and `Element`, so `import type { CheerioAPI } from 'crawlee'` (or from `@crawlee/basic` / `@crawlee/puppeteer` / ...) breaks - import them from `cheerio` and `domhandler`, which are the packages that own them. - **`@crawlee/core` no longer re-exports the internal helpers:** `parseArgument`, `schemas` and `tryAbsoluteURL` reached `@crawlee/core` (and through it `@crawlee/basic`, `@crawlee/http`, `@crawlee/browser` and `crawlee`) as public exports, which put symbols from the no-semver `/internal` entry point back into a semver-stable surface. Import them from `@crawlee/utils/internal` instead. `ArgumentValidationError` is unaffected and stays exported from `@crawlee/core`. +- **`parseSitemap`, `parseArgument` and `expandShadowRoots` are no longer on the main entry:** `parseSitemap()` (together with the `SitemapUrl` type) moved to `@crawlee/utils/internal`; use the documented `Sitemap.load()` / `Sitemap.fromXmlString()` / `Sitemap.tryCommonNames()` statics, or `discoverValidSitemaps()`, which stay on `@crawlee/utils`. `parseArgument()` was reaching the main entry through a wildcard re-export and is now only on `@crawlee/utils/internal` (`ArgumentValidationError` is unaffected and stays public). `expandShadowRoots()` moved there too — it is a DOM function that is serialized into a browser page, not a Node helper. Because the `crawlee` meta-package re-exports `@crawlee/utils` wholesale, `import { parseSitemap } from 'crawlee'` (and the same for `parseArgument` / `expandShadowRoots`) breaks as well. #### `RobotsTxtFile.find` signature changed; sitemap options removed @@ -2161,7 +2322,7 @@ const robots = await RobotsTxtFile.find(url, proxyUrl, { timeoutMillis: 5000 }); const robots = await RobotsTxtFile.find(url, { proxyUrl, timeoutMillis: 5000 }); ``` -Relatedly, `RobotsTxtFile.getSitemaps()`, `parseSitemaps()`, and `parseUrlsFromSitemaps()` no longer take a `RobotsTxtFileSitemapsOptions` argument (the type is removed), and the `enqueueStrategy` / `networkTimeouts` options were dropped from `ParseSitemapOptions` — robots/sitemap parsing no longer filters by enqueue strategy. +Relatedly, the `networkTimeouts` option was dropped from `ParseSitemapOptions`; use the single `timeoutMillis` option instead. `RobotsTxtFile.getSitemaps()`, `parseSitemaps()` and `parseUrlsFromSitemaps()` still accept an optional `RobotsTxtFileSitemapsOptions` bag, whose `enqueueStrategy` option (default `'same-hostname'`) keeps only sitemap URLs on the robots.txt host — pass `'all'` to disable that filtering. Non-`http(s)` sitemap URLs are always dropped. ### HTML-parsing helper functions are now asynchronous @@ -2191,6 +2352,45 @@ The crawler-only parts of `@crawlee/core` moved to `@crawlee/basic`, so that `@c - the crawler-only error classes: `RetryRequestError`, `RequestThrottledError`, `PersistentRateLimitError`, `NavigationSkippedError`, `MissingSessionError`, `MissingRouteError`, `RequestHandlerError` and the `ContextPipeline*Error` types `@crawlee/basic` re-exports everything from `@crawlee/core`, so `import { SessionPool } from '@crawlee/basic'` (or from `crawlee`, `@crawlee/http`, `@crawlee/playwright`, …) keeps working unchanged. Only imports written against `@crawlee/core` itself need to be pointed at `@crawlee/basic`. +### The `utils` bag is removed from the `crawlee` meta-package + +The `crawlee` meta-package exported a `utils` object — the last remnant of v2's `Apify.utils` namespace — bundling `utils.puppeteer`, `utils.playwright`, `utils.log`, `utils.social`, `utils.sleep`, `utils.downloadListOfUrls` and `utils.parseOpenGraph`. It is gone. Every member was already exported from `crawlee` under its own name, so the fix is to import that name directly: + +**Before:** +```typescript +import { utils } from 'crawlee'; + +await utils.puppeteer.saveSnapshot(page); +await utils.playwright.blockRequests(page); +utils.log.info('hello'); +await utils.sleep(1000); +const emails = utils.social.emailsFromText(text); +const urls = await utils.downloadListOfUrls({ url }); +const og = await utils.parseOpenGraph(html); +``` + +**After:** +```typescript +import { + puppeteerUtils, + playwrightUtils, + log, + sleep, + social, + downloadListOfUrls, + parseOpenGraph, +} from 'crawlee'; + +await puppeteerUtils.saveSnapshot(page); +await playwrightUtils.blockRequests(page); +log.info('hello'); +await sleep(1000); +const emails = social.emailsFromText(text); +const urls = await downloadListOfUrls({ url }); +const og = await parseOpenGraph(html); +``` + +Inside a request handler you usually do not need the namespaces at all — `saveSnapshot`, `blockRequests`, `parseWithCheerio`, `infiniteScroll` and friends are already on the crawling context, pre-bound to the current page. ## Only if you use `StagehandCrawler` @@ -2202,6 +2402,14 @@ A few Stagehand-specific option types were tightened: - The explicit `failedRequestHandler` field was removed from `StagehandCrawlerOptions` (it is inherited from the base crawler options generically, so passing `failedRequestHandler` still works). - The `ignoreShadowRoots` and `ignoreIframes` options were removed from `StagehandCrawler`. +### Removed Stagehand exports and options + +- The `StagehandRequestHandler` type was removed. It was never referenced by `StagehandCrawlerOptions.requestHandler`, which uses `RequestHandler` — use that instead. +- The `stagehandUtils` namespace was removed. Its only member was internal glue that was never part of the documented surface. +- The `AgentResult` re-export was removed. Import it from `@browserbasehq/stagehand` directly — it is a non-optional peer dependency, so it is already installed. +- `StagehandLaunchContext.stagehandOptions` was removed. It never had any effect: the value was always overwritten by the `stagehandOptions` option on the crawler and on `stagehandBrowserPool()` / `remoteStagehandBrowserPool()`. Pass `stagehandOptions` at the top level instead. +- `StagehandPlugin.stagehandOptions` is now private and `StagehandPlugin.getStagehandForBrowser()` is gone. Reach the `Stagehand` instance through the crawling context's `stagehand` property. + ## Appendix: removed symbols The full list of removed exports and members, for ctrl-F purposes. Where a replacement exists, it is noted inline. @@ -2227,7 +2435,7 @@ The full list of removed exports and members, for ctrl-F purposes. Where a repla - `Snapshotter._snapshotMemory`, `Snapshotter._memoryOverloadWarning`, `Snapshotter._snapshotEventLoop`, `Snapshotter._snapshotCpu`, `Snapshotter._snapshotClient`, `Snapshotter._pruneSnapshots` (all `@deprecated` protected stubs) - snapshotting is now handled by the individual load signals, and the `Snapshotter` itself is internal to `ConcurrencySystem`; there is no longer a public API for reading raw resource snapshots - `FileDownloadOptions.streamHandler` - streaming should now be handled directly in the `requestHandler` instead - `playwrightUtils.registerUtilsToContext` and `puppeteerUtils.registerUtilsToContext` - this is now added to the context via `ContextPipeline` composition -- `context.blockResources` and `context.cacheResponses` — no longer attached to the crawling context. The functionality is still available as deprecated functions, accessible both via the `puppeteerUtils` namespace (`puppeteerUtils.blockResources`, `puppeteerUtils.cacheResponses`) and as top-level exports from `@crawlee/puppeteer` (`import { blockResources, cacheResponses } from '@crawlee/puppeteer'`). Unlike the old context helpers, these take an explicit `page` argument — e.g. `await blockResources(page)`. Both are `@deprecated` and will be removed in a future release, so migrate away from them. +- `context.blockResources` and `context.cacheResponses`, and the `puppeteerUtils.blockResources` / `puppeteerUtils.cacheResponses` functions behind them — both had a severe performance cost in recent Puppeteer versions and were already deprecated. Use `puppeteerUtils.blockRequests(page, options)`, which blocks URL patterns over CDP without disabling the browser cache. If you were caching responses, rely on the in-browser cache instead. - `context.closeCookieModals`, `playwrightUtils.closeCookieModals` and `puppeteerUtils.closeCookieModals` — removed along with the optional `idcac-playwright` peer dependency (see [Crawling context no longer includes `closeCookieModals`](#crawling-context-no-longer-includes-closecookiemodals) and the [cookie modals guide](../guides/cookie-modals)) - `Configuration.systemInfoV2` / `CRAWLEE_SYSTEM_INFO_V2` environment variable — the v2 behavior is now the default (see [Available resource detection](#available-resource-detection)) - `Configuration.defaultDatasetId` / `defaultKeyValueStoreId` / `defaultRequestQueueId` and their `CRAWLEE_DEFAULT_*_ID` environment variables — the default storage is addressed by a reserved alias, not by a configurable ID. Open a storage by name if you need a specific one. @@ -2237,6 +2445,21 @@ The full list of removed exports and members, for ctrl-F purposes. Where a repla - `StreamHandlerContext` and `FileDownloadOptions` types (from `@crawlee/http`) — see [`FileDownload` now extends `BasicCrawler`](#filedownload-now-extends-basiccrawler-and-no-longer-takes-filedownloadoptions) - `PlainResponse` type (from `@crawlee/http`) — it wrapped the `got-scraping` response and is gone along with the rest of the old HTTP response surface (see [`CrawlingContext.response` is now of type `Response`](#crawlingcontextresponse-is-now-of-type-response)) - `checkStorageAccess`, `withCheckedStorageAccess` and the `RequestHandlerResult` type — superseded by the storage transaction mechanism; use `withDirectStorageAccess()` and `StorageTransactionView` (see [Storage writes in request handlers are transactional](#storage-writes-in-request-handlers-are-transactional)) +- `CreateContextOptions` type (from `@crawlee/basic`) — a leftover of the pre-`ContextPipeline` context-creation design, unused by the library itself; context construction is now driven by `ContextPipeline` +- `BasicCrawler.basicContextPipeline` (public getter) — the basic half of the pipeline is an implementation detail of `BasicCrawler.run()`; compose behavior via `contextPipelineBuilder` / `ContextPipeline` composition instead +- `ResponseLike` interface (from `@crawlee/core`) — a vestige of the pre-`fetch` HTTP implementation with no consumers; `getCookiesFromResponse()` has always taken a native `Response` +- `UrlPatternObject` (from `@crawlee/core`) — the *compiled* form of a URL pattern, produced internally by the `enqueueLinks()` machinery. Keep using `UrlPatternInput` / `GlobInput` / `RegExpInput`, which are unchanged, and let the return type of the pattern helpers be inferred +- `PERSIST_STATE_KEY` (from `@crawlee/core`) — to change where a session pool persists its state, pass `persistStateKey` to `SessionPool` +- `MAX_POOL_SIZE` constant (from `@crawlee/core`) — was the internal default for `SessionPoolOptions.maxPoolSize` (1000); inline the literal if you were reading it +- `WithRequired` type (from `@crawlee/core`) — a bare TypeScript utility that was never crawlee vocabulary; `LoadedRequest` no longer goes through it, so declare your own if you were using it +- `ErrorSnapshotter` and its `SnapshotResult` return type (from `@crawlee/core`) — an implementation detail of `ErrorTracker`. Error snapshotting is opt-in through `new Statistics({ saveErrorSnapshots: true })` (or `new ErrorTracker({ saveErrorSnapshots: true })`) +- `ErrorTracker.errorSnapshotter` and `ErrorTracker.captureSnapshot()` — both private now. Snapshotting is driven from `addAsync()` on the first occurrence of each distinct error; the captured URLs surface as `firstErrorScreenshotUrl` / `firstErrorHtmlUrl` on the corresponding node of `errorTracker.result`, as before +- `assertBrowserPoolNotConfigured` (from `@crawlee/browser`) — an internal helper that produced the "cannot be combined with `browserPool`" error message; it moved to `@crawlee/utils/internal` +- `MinimumSpeedStream` and `ByteCounterStream` (from `@crawlee/http`) — these `Transform` factories existed only to be piped inside `FileDownloadOptions.streamHandler`, which v4 removed. Compose your own `Transform` around `context.response.body` in the `requestHandler` instead; see the [file download with streams example](https://crawlee.dev/js/docs/examples/file-download-stream) +- `HttpHook`, `FileDownloadHook`, `CheerioHook`, `JSDOMHook` and `LinkeDOMHook` types — see [Removed navigation hook type aliases](#removed-navigation-hook-type-aliases) +- The `puppeteerClickElements` namespace (from `@crawlee/puppeteer`) — `clickElements`, `clickElementsAndInterceptNavigationRequests` and `isTargetRelevant` were internal helpers. Use `puppeteerUtils.enqueueLinksByClickingElements()`, or `context.enqueueLinksByClickingElements()` inside a request handler; the `EnqueueLinksByClickingElementsOptions` type is still exported directly from `@crawlee/puppeteer` +- The `puppeteerRequestInterception` namespace (from `@crawlee/puppeteer`) — it only duplicated `puppeteerUtils.addInterceptRequestHandler` / `puppeteerUtils.removeInterceptRequestHandler`, which are unchanged. The `InterceptHandler` type is still exported directly from `@crawlee/puppeteer` +- The top-level type exports `BlockRequestsOptions`, `InjectFileOptions`, `InfiniteScrollOptions`, `SaveSnapshotOptions`, `CompiledScriptParams` and `CompiledScriptFunction` (from `@crawlee/puppeteer`) — the types themselves are unchanged and remain reachable through the namespace, e.g. `import { puppeteerUtils } from 'crawlee'; let options: puppeteerUtils.SaveSnapshotOptions;`. This matches `@crawlee/playwright`, which never exported them at the top level. `PuppeteerDirectNavigationOptions` is unaffected #### The protected `BasicCrawler.crawlingContexts` map is removed diff --git a/packages/basic-crawler/src/internals/autoscaling/concurrency_system.ts b/packages/basic-crawler/src/internals/autoscaling/concurrency_system.ts index 8cedfcb5ed32..bc840e7b5b86 100644 --- a/packages/basic-crawler/src/internals/autoscaling/concurrency_system.ts +++ b/packages/basic-crawler/src/internals/autoscaling/concurrency_system.ts @@ -331,21 +331,14 @@ export class ConcurrencySystem implements IConcurrencySystem { /** * Gets the desired concurrency for the system, * which is an estimated number of parallel tasks that the system can currently support. + * + * Read-only: the value is owned by the autoscaler, which recomputes it from the load signals on every tick. + * Tune the budget through `minConcurrency`/`maxConcurrency` instead. */ get desiredConcurrency(): number { return this.#desiredConcurrency; } - /** - * Sets the desired concurrency for the system, i.e. the number of tasks that should be running - * in parallel if there's large enough supply of tasks. - */ - set desiredConcurrency(value: number) { - parseArgument(value, concurrencySchema); - this.#desiredConcurrency = value; - this.clampDesiredConcurrency(); - } - /** * Re-establishes `minConcurrency <= desiredConcurrency <= maxConcurrency` after any of the three is retuned. * Dispatch gates on the desired value alone, so one stranded above `maxConcurrency` would make the ceiling diff --git a/packages/basic-crawler/src/internals/basic-crawler.ts b/packages/basic-crawler/src/internals/basic-crawler.ts index 5e3f6c832956..f2cfe4fd4844 100644 --- a/packages/basic-crawler/src/internals/basic-crawler.ts +++ b/packages/basic-crawler/src/internals/basic-crawler.ts @@ -49,7 +49,6 @@ import type { Dictionary, ISession, ISessionPool, - ProxyInfo, SetStatusMessageOptions, StorageBackend, } from '@crawlee/types'; @@ -83,7 +82,7 @@ import { } from './errors.js'; import type { IStatistics, StatisticState } from './crawlers/statistics.js'; import { Statistics } from './crawlers/statistics.js'; -import type { EnqueueUrlsOptions, SkippedRequestCallback, UrlPatternObject } from './enqueue_links/index.js'; +import type { EnqueueUrlsOptions, SkippedRequestCallback } from './enqueue_links/index.js'; import { applyRequestTransform, buildEnqueueStrategyPatterns, @@ -819,18 +818,16 @@ export class BasicCrawler< * pipelines expect the basic crawler fields to already be present in the context at runtime. * * Context built with this pipeline can be passed into multiple crawler pipelines at once. - * This is used e.g. in the {@apilink AdaptivePlaywrightCrawler|`AdaptivePlaywrightCrawler`}. */ - get basicContextPipeline(): ContextPipeline<{ request: Request }, CrawlingContext> { - if (this.#basicContextPipeline === undefined) { - this.#basicContextPipeline = this.buildBasicContextPipeline(); - } + get #basicPipeline(): ContextPipeline<{ request: Request }, CrawlingContext> { + this.#basicContextPipeline ??= this.buildBasicContextPipeline(); return this.#basicContextPipeline; } #contextPipeline?: ContextPipeline; + /** @internal */ get contextPipeline(): ContextPipeline { if (this.#contextPipeline === undefined) { this.#contextPipeline = this.buildFinalContextPipeline(); @@ -1248,7 +1245,7 @@ export class BasicCrawler< // catch-all for that - see `raceWithTimeout` for why it is a bare timer, not a timeout frame. await this.withRequestTimeout( crawlingContext, - this.basicContextPipeline + this.#basicPipeline .chain(this.contextPipeline) .call(crawlingContext, (ctx) => this.handleRequest(ctx, source, request)), ), @@ -1599,7 +1596,7 @@ export class BasicCrawler< * @param error The error to check. */ protected isProxyError(error: Error): boolean { - return ROTATE_PROXY_ERRORS.some((x: string) => (this.getMessageFromError(error) as any)?.includes(x)); + return ROTATE_PROXY_ERRORS.some((x: string) => this.getMessageFromError(error).includes(x)); } /** @@ -2112,17 +2109,11 @@ export class BasicCrawler< const requestLimit = await this.#calculateEnqueuedRequestLimit(options.limit); const strategy = options.strategy ?? EnqueueStrategy.All; - const urlExcludePatternObjects: UrlPatternObject[] = options.exclude?.length - ? constructUrlPatternObjects(options.exclude) - : []; - const urlPatternObjects: UrlPatternObject[] = options.include?.length - ? constructUrlPatternObjects(options.include) - : []; + const urlExcludePatternObjects = options.exclude?.length ? constructUrlPatternObjects(options.exclude) : []; + const urlPatternObjects = options.include?.length ? constructUrlPatternObjects(options.include) : []; // The strategy always applies, even when `include` patterns are provided - the two are AND-ed together // (a URL must match an `include` pattern *and* satisfy the strategy). This mirrors crawlee-python. - const enqueueStrategyPatterns: UrlPatternObject[] = options.baseUrl - ? buildEnqueueStrategyPatterns(options.baseUrl, strategy) - : []; + const enqueueStrategyPatterns = options.baseUrl ? buildEnqueueStrategyPatterns(options.baseUrl, strategy) : []; const isAllowedBasedOnRobotsTxtFile = this.isAllowedBasedOnRobotsTxtFile.bind(this); const maxCrawlDepth = this.#maxCrawlDepth; @@ -2566,6 +2557,7 @@ export class BasicCrawler< ); } + /** @internal */ protected async getRobotsTxtFileForUrl(url: string): Promise { if (!this.#respectRobotsTxtFile) { return undefined; @@ -2896,7 +2888,7 @@ export class BasicCrawler< * @param error The error received * @returns The message to be logged */ - protected getMessageFromError(error: Error, forceStack = false) { + protected getMessageFromError(error: Error, forceStack = false): string { if ([TypeError, SyntaxError, ReferenceError].some((type) => error instanceof type)) { forceStack = true; } @@ -2907,7 +2899,8 @@ export class BasicCrawler< const userLine = stackLines.find((line) => line.includes(baseDir) && !line.includes('node_modules')); if (error instanceof TimeoutError) { - return process.env.CRAWLEE_VERBOSE_LOG ? error.stack : error.message || error; // stack in timeout errors does not really help + // stack in timeout errors does not really help + return (process.env.CRAWLEE_VERBOSE_LOG && error.stack) || error.message; } return process.env.CRAWLEE_VERBOSE_LOG || forceStack @@ -3054,12 +3047,6 @@ export class BasicCrawler< } } -export interface CreateContextOptions { - request: Request; - session: ISession; - proxyInfo?: ProxyInfo; -} - export interface CrawlerAddRequestsOptions extends AddRequestsBatchedOptions, EnqueueUrlsOptions {} export interface CrawlerAddRequestsResult extends AddRequestsBatchedResult {} diff --git a/packages/basic-crawler/src/internals/cookie_utils.ts b/packages/basic-crawler/src/internals/cookie_utils.ts index 43286b7aa10a..d4559f3762bf 100644 --- a/packages/basic-crawler/src/internals/cookie_utils.ts +++ b/packages/basic-crawler/src/internals/cookie_utils.ts @@ -4,11 +4,6 @@ import { Cookie, CookieJar } from 'tough-cookie'; import { CookieParseError } from './session_pool/errors.js'; -export interface ResponseLike { - url?: string | (() => string); - headers?: Record | (() => Record); -} - /** * @internal */ diff --git a/packages/basic-crawler/src/internals/crawlers/crawler_commons.ts b/packages/basic-crawler/src/internals/crawlers/crawler_commons.ts index 3cc5fba4a01b..5448f132060f 100644 --- a/packages/basic-crawler/src/internals/crawlers/crawler_commons.ts +++ b/packages/basic-crawler/src/internals/crawlers/crawler_commons.ts @@ -91,9 +91,11 @@ export type TypedContextEnqueueLinks< ? (options: TypedEnqueueLinksOptions) => Result : EnqueueLinks; -export type WithRequired = T & { [P in K]-?: T[P] }; - -export type LoadedRequest = WithRequired; +/** A {@apilink Request} that has been dispatched, so its `id` and `loadedUrl` are guaranteed to be present. */ +// `Required>` rather than an inline `{ [P in 'id' | 'loadedUrl']-?: R[P] }`: only a homomorphic +// mapped type (one keyed by `keyof X`, as `Required` is) strips `undefined` from the property type. Keyed +// by a literal union it would merely drop the `?`, leaving `string | undefined`. +export type LoadedRequest = R & Required>; /** @internal */ export type LoadedContext = @@ -104,6 +106,7 @@ export type LoadedContext = } & Omit; export interface RestrictedCrawlingContext { + /** @internal */ id: string; session: ISession; diff --git a/packages/basic-crawler/src/internals/crawlers/error_snapshotter.ts b/packages/basic-crawler/src/internals/crawlers/error_snapshotter.ts index 4837a6fd0bdc..04ad63ce4f3f 100644 --- a/packages/basic-crawler/src/internals/crawlers/error_snapshotter.ts +++ b/packages/basic-crawler/src/internals/crawlers/error_snapshotter.ts @@ -11,7 +11,7 @@ interface BrowserCrawlingContext { saveSnapshot: (options: { key: string }) => Promise; } -export interface SnapshotResult { +interface SnapshotResult { screenshotFileName?: string; htmlFileName?: string; } @@ -23,6 +23,12 @@ interface ErrorSnapshot { htmlFileUrl?: string; } +const MAX_ERROR_CHARACTERS = 30; +const MAX_HASH_LENGTH = 30; +const MAX_FILENAME_LENGTH = 250; +const BASE_MESSAGE = 'An error occurred'; +const SNAPSHOT_PREFIX = 'ERROR_SNAPSHOT'; + /** * ErrorSnapshotter class is used to capture a screenshot of the page and a snapshot of the HTML when an error occurs during web crawling. * @@ -36,12 +42,6 @@ interface ErrorSnapshot { * ``` */ export class ErrorSnapshotter { - static readonly MAX_ERROR_CHARACTERS = 30; - static readonly MAX_HASH_LENGTH = 30; - static readonly MAX_FILENAME_LENGTH = 250; - static readonly BASE_MESSAGE = 'An error occurred'; - static readonly SNAPSHOT_PREFIX = 'ERROR_SNAPSHOT'; - /** * Capture a snapshot of the error context. */ @@ -59,13 +59,13 @@ export class ErrorSnapshotter { return {}; } - const fileName = this.generateFilename(error); + const fileName = this.#generateFilename(error); let screenshotFileName: string | undefined; let htmlFileName: string | undefined; if (page) { - const capturedFiles = await this.contextCaptureSnapshot( + const capturedFiles = await this.#contextCaptureSnapshot( context as unknown as BrowserCrawlingContext, fileName, ); @@ -78,11 +78,11 @@ export class ErrorSnapshotter { // If the snapshot for browsers failed to capture the HTML, try to capture it from the page content if (!htmlFileName) { const html = await page.content(); - htmlFileName = html ? await this.saveHTMLSnapshot(html, keyValueStore, fileName) : undefined; + htmlFileName = html ? await this.#saveHTMLSnapshot(html, keyValueStore, fileName) : undefined; } } else if (typeof body === 'string') { // for non-browser contexts - htmlFileName = await this.saveHTMLSnapshot(body, keyValueStore, fileName); + htmlFileName = await this.#saveHTMLSnapshot(body, keyValueStore, fileName); } return { @@ -101,7 +101,7 @@ export class ErrorSnapshotter { * This function is applicable for browser contexts only. * Returns an object containing the filenames of the screenshot and HTML file. */ - async contextCaptureSnapshot( + async #contextCaptureSnapshot( context: BrowserCrawlingContext, fileName: string, ): Promise { @@ -119,7 +119,7 @@ export class ErrorSnapshotter { /** * Save the HTML snapshot of the page, and return the key it was stored under. */ - async saveHTMLSnapshot( + async #saveHTMLSnapshot( html: string, keyValueStore: Pick, fileName: string, @@ -137,9 +137,7 @@ export class ErrorSnapshotter { /** * Generate a unique fileName for each error snapshot. */ - generateFilename(error: ErrnoException): string { - const { SNAPSHOT_PREFIX, BASE_MESSAGE, MAX_HASH_LENGTH, MAX_ERROR_CHARACTERS, MAX_FILENAME_LENGTH } = - ErrorSnapshotter; + #generateFilename(error: ErrnoException): string { // Create a hash of the error stack trace const errorStackHash = crypto .createHash('sha1') diff --git a/packages/basic-crawler/src/internals/crawlers/error_tracker.ts b/packages/basic-crawler/src/internals/crawlers/error_tracker.ts index fa085a188a64..48faa6d4c390 100644 --- a/packages/basic-crawler/src/internals/crawlers/error_tracker.ts +++ b/packages/basic-crawler/src/internals/crawlers/error_tracker.ts @@ -287,11 +287,21 @@ const increaseCount = (group: { count?: number }) => { export class ErrorTracker { #options: ErrorTrackerOptions; - result: Record; + #result: Record; - total: number; + #total: number; - errorSnapshotter?: ErrorSnapshotter; + #errorSnapshotter?: ErrorSnapshotter; + + /** The grouped error tree the tracker accumulates. Read-only: the object is mutated in place as errors arrive. */ + get result(): Record { + return this.#result; + } + + /** The total number of errors passed to the tracker, including the same error seen repeatedly. */ + get total(): number { + return this.#total; + } constructor(options: Partial = {}) { this.#options = { @@ -306,15 +316,15 @@ export class ErrorTracker { }; if (this.#options.saveErrorSnapshots) { - this.errorSnapshotter = new ErrorSnapshotter(); + this.#errorSnapshotter = new ErrorSnapshotter(); } - this.result = Object.create(null); - this.total = 0; + this.#result = Object.create(null); + this.#total = 0; } private updateGroup(error: ErrnoException) { - let group = this.result; + let group = this.#result; if (this.#options.showStackTrace) { group = getStackTraceGroup(error, group, this.#options.showFullStack); @@ -338,7 +348,7 @@ export class ErrorTracker { } add(error: ErrnoException) { - this.total++; + this.#total++; this.updateGroup(error); @@ -352,13 +362,13 @@ export class ErrorTracker { * We added this new method to avoid breaking changes. */ async addAsync(error: ErrnoException, context?: CrawlingContext) { - this.total++; + this.#total++; const group = this.updateGroup(error); // Capture a snapshot (screenshot and HTML) on the first occurrence of an error if (group.count === 1 && context) { - await this.captureSnapshot(group, error, context).catch(() => {}); + await this.#captureSnapshot(group, error, context).catch(() => {}); } if (typeof error.cause === 'object' && error.cause !== null) { @@ -381,7 +391,7 @@ export class ErrorTracker { } }; - goDeeper(this.result); + goDeeper(this.#result); return count; } @@ -401,21 +411,21 @@ export class ErrorTracker { } }; - goDeeper(this.result, []); + goDeeper(this.#result, []); return result.sort((a, b) => b[0] - a[0]).slice(0, count); } - async captureSnapshot( + async #captureSnapshot( storage: Record, error: ErrnoException, context: CrawlingContext & SnapshottableProperties, ) { - if (!this.errorSnapshotter) { + if (!this.#errorSnapshotter) { return; } - const { screenshotFileUrl, htmlFileUrl } = await this.errorSnapshotter.captureSnapshot(error, context); + const { screenshotFileUrl, htmlFileUrl } = await this.#errorSnapshotter.captureSnapshot(error, context); storage.firstErrorScreenshotUrl = screenshotFileUrl; storage.firstErrorHtmlUrl = htmlFileUrl; @@ -424,8 +434,8 @@ export class ErrorTracker { reset() { // This actually safe, since we Object.create(null) so no prototype pollution can happen. // eslint-disable-next-line no-restricted-syntax, guard-for-in - for (const key in this.result) { - delete this.result[key]; + for (const key in this.#result) { + delete this.#result[key]; } } } diff --git a/packages/basic-crawler/src/internals/crawlers/index.ts b/packages/basic-crawler/src/internals/crawlers/index.ts index 9a101845482d..ad002da1057f 100644 --- a/packages/basic-crawler/src/internals/crawlers/index.ts +++ b/packages/basic-crawler/src/internals/crawlers/index.ts @@ -2,4 +2,3 @@ export * from './context_pipeline.js'; export type * from './crawler_commons.js'; export * from './statistics.js'; export * from './error_tracker.js'; -export * from './error_snapshotter.js'; diff --git a/packages/basic-crawler/src/internals/enqueue_links/enqueue_links.ts b/packages/basic-crawler/src/internals/enqueue_links/enqueue_links.ts index 1b146ba644ed..ce95352617f2 100644 --- a/packages/basic-crawler/src/internals/enqueue_links/enqueue_links.ts +++ b/packages/basic-crawler/src/internals/enqueue_links/enqueue_links.ts @@ -1,5 +1,5 @@ import type { Dictionary } from '@crawlee/types'; -import { EnqueueStrategy } from '@crawlee/utils'; +import { EnqueueStrategy } from '@crawlee/utils/internal'; import { getDomain } from 'tldts'; import type { EnqueueStrategyOption, RequestQueueOperationOptions } from '@crawlee/core'; diff --git a/packages/basic-crawler/src/internals/enqueue_links/index.ts b/packages/basic-crawler/src/internals/enqueue_links/index.ts index 3582f2a5eb7d..867f10e2eca4 100644 --- a/packages/basic-crawler/src/internals/enqueue_links/index.ts +++ b/packages/basic-crawler/src/internals/enqueue_links/index.ts @@ -1,2 +1,24 @@ export * from './enqueue_links.js'; -export * from './shared.js'; +// Not `export *`: `UrlPatternObject` is the compiled internal form of a `UrlPatternInput` and carries no semver +// guarantees. Internal consumers import it from `./shared.js` directly. +export { + applyRequestTransform, + constructGlobObjectsFromGlobs, + constructRegExpObjectsFromRegExps, + constructUrlPatternObjects, + createRequestOptions, + createSkippedRequestArgs, + filterRequestOptionsByPatterns, + updateEnqueueLinksPatternCache, + urlPatternSchema, + validateGlobPattern, +} from './shared.js'; +export type { + GlobInput, + GlobObject, + RegExpInput, + RegExpObject, + RequestTransform, + SkippedRequestCallback, + UrlPatternInput, +} from './shared.js'; diff --git a/packages/basic-crawler/src/internals/enqueue_links/shared.ts b/packages/basic-crawler/src/internals/enqueue_links/shared.ts index 16274c7bfa50..4c16e1fb8a22 100644 --- a/packages/basic-crawler/src/internals/enqueue_links/shared.ts +++ b/packages/basic-crawler/src/internals/enqueue_links/shared.ts @@ -17,11 +17,6 @@ const MAX_ENQUEUE_LINKS_CACHE_SIZE = 1000; */ const enqueueLinksPatternCache = new Map(); -export interface UrlPatternObject { - glob?: string; - regexp?: RegExp; -} - export interface GlobObject { glob: string; } @@ -162,6 +157,16 @@ export function constructRegExpObjectsFromRegExps(regexps: readonly RegExpInput[ }); } +/** + * The compiled form of a {@apilink UrlPatternInput}, produced by {@apilink constructUrlPatternObjects}. Deliberately + * not re-exported from the package entry point — it is an implementation detail of the pattern matching helpers. + * @internal + */ +export interface UrlPatternObject { + glob?: string; + regexp?: RegExp; +} + /** * Helper factory used in the `enqueueLinks()` function to construct UrlPatternObjects * from a mixed array of glob strings, glob objects, RegExp instances, and regexp objects. diff --git a/packages/basic-crawler/src/internals/router.ts b/packages/basic-crawler/src/internals/router.ts index 75d754a23651..0132b176b6b9 100644 --- a/packages/basic-crawler/src/internals/router.ts +++ b/packages/basic-crawler/src/internals/router.ts @@ -379,6 +379,7 @@ export class Router< * override it and the crawler's own timeout should apply. Falls back to the default route the same way * {@apilink Router.getHandler|`getHandler`} does, so a label with no route of its own inherits whatever * the default route asked for. Used by the crawler; not meant to be called directly. + * @internal */ getTimeoutSecs(label?: string | symbol): number | undefined { if (label && this.#routes.has(label)) { @@ -391,6 +392,7 @@ export class Router< /** * The longest `requestHandlerTimeoutSecs` any route asked for, or `undefined` when no route overrides it. * The crawler needs an upper bound up front, before it knows which routes a run will actually hit. + * @internal */ getMaxTimeoutSecs(): number | undefined { return this.#timeouts.size > 0 ? Math.max(...this.#timeouts.values()) : undefined; diff --git a/packages/basic-crawler/src/internals/session_pool/consts.ts b/packages/basic-crawler/src/internals/session_pool/consts.ts index 7fbe1ed1b510..a4aaed1fd2f3 100644 --- a/packages/basic-crawler/src/internals/session_pool/consts.ts +++ b/packages/basic-crawler/src/internals/session_pool/consts.ts @@ -1,3 +1 @@ -export const BLOCKED_STATUS_CODES = [401, 403, 429]; -export const PERSIST_STATE_KEY = 'CRAWLEE_SESSION_POOL_STATE'; -export const MAX_POOL_SIZE = 1000; +export const BLOCKED_STATUS_CODES: readonly number[] = [401, 403, 429]; diff --git a/packages/basic-crawler/src/internals/session_pool/session.ts b/packages/basic-crawler/src/internals/session_pool/session.ts index e048e06fc474..f27412fda272 100644 --- a/packages/basic-crawler/src/internals/session_pool/session.ts +++ b/packages/basic-crawler/src/internals/session_pool/session.ts @@ -86,6 +86,7 @@ export interface SessionOptions { */ retired?: boolean; + /** @internal */ log?: CrawleeLogger; errorScore?: number; cookieJar?: CookieJar; @@ -264,6 +265,7 @@ export class Session implements ISession { /** * Gets session state for persistence in KeyValueStore. + * @internal */ getState(): SessionState { diff --git a/packages/basic-crawler/src/internals/session_pool/session_pool.ts b/packages/basic-crawler/src/internals/session_pool/session_pool.ts index 66916e0554b4..614c2d3b750d 100644 --- a/packages/basic-crawler/src/internals/session_pool/session_pool.ts +++ b/packages/basic-crawler/src/internals/session_pool/session_pool.ts @@ -6,13 +6,26 @@ import { AsyncQueue } from '@sapphire/async-queue'; import { z } from 'zod'; import type { PersistenceOptions } from '../crawlers/statistics.js'; -import { MAX_POOL_SIZE, PERSIST_STATE_KEY } from './consts.js'; import { createDefaultSessionFingerprint } from './fingerprint.js'; import type { SessionOptions } from './session.js'; import { Session } from './session.js'; -const SESSION_REUSE_STRATEGIES = ['random', 'round-robin', 'use-until-failure'] as const; -export type SessionReuseStrategy = (typeof SESSION_REUSE_STRATEGIES)[number]; +/** Default upper bound on the number of sessions the pool keeps around. */ +const MAX_POOL_SIZE = 1000; + +/** Prefix of the default key the pool persists its state under; the pool id is appended to it. */ +const PERSIST_STATE_KEY = 'CRAWLEE_SESSION_POOL_STATE'; + +export type SessionReuseStrategy = 'random' | 'round-robin' | 'use-until-failure'; + +// Runtime mirror of `SessionReuseStrategy` for the zod schema below. Declared this way round - and not as +// `typeof SESSION_REUSE_STRATEGIES[number]` - so the public type is a plain literal union that does not drag +// the array into the API surface. `satisfies` keeps the array from naming a strategy the type does not have. +const SESSION_REUSE_STRATEGIES = [ + 'random', + 'round-robin', + 'use-until-failure', +] as const satisfies readonly SessionReuseStrategy[]; // `schemas.anyObject` passes values through by reference (object schemas return a pruned plain // copy), so class instances like loggers keep their prototype. diff --git a/packages/basic-crawler/src/internals/sitemap_request_loader.ts b/packages/basic-crawler/src/internals/sitemap_request_loader.ts index abc5ef252869..94f1bada133c 100644 --- a/packages/basic-crawler/src/internals/sitemap_request_loader.ts +++ b/packages/basic-crawler/src/internals/sitemap_request_loader.ts @@ -1,7 +1,8 @@ import { Transform } from 'node:stream'; import type { BaseHttpClient } from '@crawlee/http-client'; -import { EnqueueStrategy, parseSitemap, type ParseSitemapOptions } from '@crawlee/utils'; +import type { ParseSitemapOptions } from '@crawlee/utils'; +import { EnqueueStrategy, parseArgument, parseSitemap, schemas } from '@crawlee/utils/internal'; import { minimatch } from 'minimatch'; import type { RequiredDeep } from 'type-fest'; import { z } from 'zod'; @@ -16,9 +17,9 @@ import { RequestQueue, serviceLocator, } from '@crawlee/core'; -import { parseArgument, schemas } from '@crawlee/utils/internal'; -import type { UrlPatternInput, UrlPatternObject } from './enqueue_links/index.js'; +import type { UrlPatternInput } from './enqueue_links/index.js'; +import type { UrlPatternObject } from './enqueue_links/shared.js'; import { constructUrlPatternObjects, urlPatternSchema } from './enqueue_links/index.js'; const sitemapRequestLoaderOptionsSchema = z.strictObject({ diff --git a/packages/basic-crawler/src/internals/throttling_request_manager.ts b/packages/basic-crawler/src/internals/throttling_request_manager.ts index a0b32de39c3f..8cc886729bc1 100644 --- a/packages/basic-crawler/src/internals/throttling_request_manager.ts +++ b/packages/basic-crawler/src/internals/throttling_request_manager.ts @@ -49,8 +49,10 @@ const throttlingRequestManagerOptionsSchema = z.strictObject({ * * {@apilink ThrottlingRequestManager} calls this once per configured domain, so every per-domain queue shares the * concrete type and storage backend of the manager being wrapped. + * + * Not exported: the only option that takes one is `@internal`. */ -export type RequestManagerOpener = ( +type RequestManagerOpener = ( identifier?: string | StorageIdentifier | null, options?: StorageOpenOptions, ) => Promise; @@ -127,6 +129,7 @@ export interface ThrottlingRequestManagerOptions`. * @default RequestQueue.open + * @internal */ requestManagerOpener?: RequestManagerOpener; @@ -396,6 +399,8 @@ export class ThrottlingRequestManager { await this.#forEachManager((manager) => (manager as { drop?(): Promise }).drop?.()); this.#subManagers.clear(); diff --git a/packages/browser-crawler/src/internals/browser-crawler.ts b/packages/browser-crawler/src/internals/browser-crawler.ts index 7722871da46f..884c9a9b0065 100644 --- a/packages/browser-crawler/src/internals/browser-crawler.ts +++ b/packages/browser-crawler/src/internals/browser-crawler.ts @@ -32,6 +32,7 @@ import { import type { CommonPage, CrawlerRemoteBrowserOptions } from '@crawlee/browser-pool'; import type { Awaitable, Cookie as CookieObject, Dictionary, IBrowserPool, ISession } from '@crawlee/types'; import { + assertBrowserPoolNotConfigured, CLOUDFLARE_RETRY_CSS_SELECTORS, parseArgument, RETRY_CSS_SELECTORS, @@ -43,8 +44,6 @@ import { z } from 'zod'; import { addTimeoutToPromise, TimeoutError, tryCancel } from '@apify/timeout'; -import type { BrowserLaunchContext } from './browser-launcher.js'; - interface BaseResponse { status(): number; /** Optional because only Playwright and Puppeteer responses are guaranteed to carry it. */ @@ -62,25 +61,6 @@ export type OwnedBrowserPool = IBrowserPool & { destroy: () => Promise; }; -/** - * Rejects options that exist only to configure the browser pool the crawler would have built for itself. - * Accepting them alongside a pre-built `browserPool` and quietly ignoring them is how `browserPoolOptions` grew - * into a second, half-working way of configuring the same pool. - */ -export function assertBrowserPoolNotConfigured(crawlerName: string, ignoredOptions: Dictionary): void { - const names = Object.keys(ignoredOptions).filter((name) => ignoredOptions[name] !== undefined); - - if (names.length === 0) { - return; - } - - throw new Error( - `${crawlerName}: ${names.map((name) => `\`${name}\``).join(', ')} cannot be combined with \`browserPool\`, ` + - `${names.length > 1 ? 'they configure' : 'it configures'} the browser pool the crawler would build for ` + - 'itself. Configure the pool you pass in instead.', - ); -} - type ContextDifference = Omit & Partial; export interface BrowserCrawlingContext< @@ -142,8 +122,6 @@ export interface BrowserCrawlerOptions< // Overridden with browser context 'requestHandler' | 'failedRequestHandler' | 'errorHandler' > { - launchContext?: BrowserLaunchContext; - /** * The browser pool the crawler should serve its pages from. This is the single way to run a pool with * non-default options: build one with the factory that matches your crawler @@ -362,7 +340,6 @@ function isNavigationTimeoutError(error: Error): boolean { export abstract class BrowserCrawler< Page extends CommonPage = CommonPage, Response extends BaseResponse = BaseResponse, - LaunchOptions extends Dictionary | undefined = Dictionary, Context extends BrowserCrawlingContext = BrowserCrawlingContext, ContextExtension = Dictionary, ExtendedContext extends Context = Context & ContextExtension, @@ -381,8 +358,6 @@ export abstract class BrowserCrawler< return this.#browserPoolDep.value; } - launchContext: BrowserLaunchContext; - protected readonly ignoreShadowRoots: boolean; protected readonly ignoreIframes: boolean; @@ -403,7 +378,6 @@ export abstract class BrowserCrawler< preNavigationHooks: schemas.anyArray.default(() => []), postNavigationHooks: schemas.anyArray.default(() => []), - launchContext: schemas.anyObject.default(() => ({})), browserPool: validators.browserPool.optional(), browserPoolBuilder: schemas.anyFunction.optional(), remoteBrowser: schemas.anyObject.optional(), @@ -418,6 +392,7 @@ export abstract class BrowserCrawler< /** * All `BrowserCrawler` parameters are passed via an options object. + * @internal */ protected constructor( options: BrowserCrawlerOptions< @@ -441,7 +416,6 @@ export abstract class BrowserCrawler< const { navigationTimeoutSecs, saveResponseCookies, - launchContext, browserPool, remoteBrowser, preNavigationHooks, @@ -508,7 +482,6 @@ export abstract class BrowserCrawler< extendContext, }); - this.launchContext = launchContext; this.#navigationTimeoutMillis = navigationTimeoutSecs * 1000; // The public option hooks are extension-aware; internal storage uses the base context type // (the pipeline composes hooks against the concrete context, which does not statically carry @@ -523,10 +496,12 @@ export abstract class BrowserCrawler< this.#browserPoolDep = OwnedOrInjected.resolve(browserPool, () => browserPoolBuilder(remoteBrowser)); } + /** @internal */ protected override getNavigationTimeoutMillis(): number { return this.#navigationTimeoutMillis; } + /** @internal */ protected override buildContextPipeline(): ContextPipeline< CrawlingContext, BrowserCrawlingContext @@ -752,6 +727,7 @@ export abstract class BrowserCrawler< /** * Runs the user request handler, then re-reads browser cookies so login flows / * `page.setCookie` / XHR `Set-Cookie` updates are stored for later requests. + * @internal */ protected override async runRequestHandler(crawlingContext: ExtendedContext): Promise { try { @@ -822,7 +798,7 @@ export abstract class BrowserCrawler< */ private throwIfProxyError(error: Error) { if (this.isProxyError(error)) { - throw new SessionError(this.getMessageFromError(error) as string); + throw new SessionError(this.getMessageFromError(error)); } } diff --git a/packages/browser-crawler/src/internals/browser-launcher.ts b/packages/browser-crawler/src/internals/browser-launcher.ts index 13dfc1cae624..fb75c3f40752 100644 --- a/packages/browser-crawler/src/internals/browser-launcher.ts +++ b/packages/browser-crawler/src/internals/browser-launcher.ts @@ -98,14 +98,14 @@ export interface BrowserLaunchContext extends BrowserPluginO * the browser they run against is only known to the concrete `*BrowserPool()` factory, which is where the * caller-facing types are pinned down. */ -export type LauncherBrowserPoolOptions = Omit & { +type LauncherBrowserPoolOptions = Omit & { [Hook in keyof BrowserPoolHooks]?: readonly ((...args: any[]) => unknown)[]; }; /** * The {@apilink RemoteBrowserPool} counterpart of {@apilink LauncherBrowserPoolOptions}. */ -export type LauncherRemoteBrowserPoolOptions = Omit; +type LauncherRemoteBrowserPoolOptions = Omit; /** * Abstract class for creating browser launchers, such as `PlaywrightLauncher` and `PuppeteerLauncher`. diff --git a/packages/browser-pool/src/abstract-classes/browser-controller.ts b/packages/browser-pool/src/abstract-classes/browser-controller.ts index d8a0c74f0a99..e58f7a8bcad8 100644 --- a/packages/browser-pool/src/abstract-classes/browser-controller.ts +++ b/packages/browser-pool/src/abstract-classes/browser-controller.ts @@ -113,6 +113,13 @@ export abstract class BrowserController< implements IBrowserController { readonly id = nanoid(); + + /** + * Kept `protected` rather than `#private` because the concrete controllers in this package are + * separate classes — an ES private field would not be visible to them. + * + * @internal + */ protected readonly log!: CrawleeLogger; /** @@ -136,6 +143,7 @@ export abstract class BrowserController< */ proxyUrl?: string; + /** @internal */ isActive = false; activePages = 0; @@ -144,8 +152,10 @@ export abstract class BrowserController< readonly #pageTeardowns = new WeakMap Promise>(); + /** @internal */ totalPages = 0; + /** @internal */ lastPageOpenedAt = Date.now(); #activate!: () => void; @@ -298,31 +308,18 @@ export abstract class BrowserController< return this._getCookies(page); } - /** - * @private - */ protected abstract _close(): Promise; - /** - * @private - */ + protected abstract _kill(): Promise; - /** - * @private - */ + protected abstract _newPage(pageOptions?: NewPageOptions): Promise; - /** - * @private - */ protected abstract _setCookies(page: NewPageResult, cookies: Cookie[]): Promise; - /** - * @private - */ protected abstract _getCookies(page: NewPageResult): Promise; /** - * @private + * @internal */ abstract normalizeProxyOptions(proxyUrl: string | undefined, pageOptions: any): Record; } diff --git a/packages/browser-pool/src/abstract-classes/browser-plugin.ts b/packages/browser-pool/src/abstract-classes/browser-plugin.ts index 0fb47e975626..4d94aeb2247f 100644 --- a/packages/browser-pool/src/abstract-classes/browser-plugin.ts +++ b/packages/browser-pool/src/abstract-classes/browser-plugin.ts @@ -105,7 +105,15 @@ export abstract class BrowserPlugin< NewPageResult = UnwrapPromise>, > { readonly name = this.constructor.name; + + /** + * Kept `protected` rather than `#private` because the concrete plugins in this package are separate + * classes — an ES private field would not be visible to them. + * + * @internal + */ protected readonly log!: CrawleeLogger; + readonly library: Library; readonly launchOptions: LibraryOptions; readonly proxyUrl?: string; @@ -320,9 +328,6 @@ export abstract class BrowserPlugin< throw new BrowserLaunchError(`${errorMessage.join('\n')}\u200b`, { cause }); } - /** - * @private - */ protected abstract addProxyToLaunchOptions( launchContext: LaunchContext, ): Promise; @@ -331,9 +336,6 @@ export abstract class BrowserPlugin< launchContext: LaunchContext, ): boolean; - /** - * @private - */ protected abstract _launch( launchContext: LaunchContext, ): Promise; diff --git a/packages/browser-pool/src/anonymize-proxy.ts b/packages/browser-pool/src/anonymize-proxy.ts index 33b7176f556f..0fb46a5639be 100644 --- a/packages/browser-pool/src/anonymize-proxy.ts +++ b/packages/browser-pool/src/anonymize-proxy.ts @@ -1,5 +1,6 @@ type PromiseVoid = () => Promise; +/** @internal */ export interface AnonymizeProxySugarOptions { ignoreProxyCertificate?: boolean; } diff --git a/packages/browser-pool/src/browser-pool.ts b/packages/browser-pool/src/browser-pool.ts index 9d9d0c6c3ee8..572efa011535 100644 --- a/packages/browser-pool/src/browser-pool.ts +++ b/packages/browser-pool/src/browser-pool.ts @@ -327,30 +327,43 @@ export class BrowserPool< implements IBrowserPool { browserPlugins: BrowserPlugins; - maxOpenPagesPerBrowser: number; + + /** @internal */ maxOpenBrowsers: number; - retireBrowserAfterPageCount: number; - operationTimeoutMillis: number; - closeInactiveBrowserAfterMillis: number; - useFingerprints?: boolean; + + /** @internal */ fingerprintOptions: FingerprintOptions; - preLaunchHooks: PreLaunchHook[]; - postLaunchHooks: PostLaunchHook[]; - prePageCreateHooks: PrePageCreateHook[]; - postPageCreateHooks: PostPageCreateHook[]; - prePageCloseHooks: PrePageCloseHook[]; - postPageCloseHooks: PostPageCloseHook[]; - pageCounter = 0; - pages = new Map(); - pageIds = new WeakMap(); - startingBrowserControllers = new Set(); - activeBrowserControllers = new Set(); - retiredBrowserControllers = new Set(); - pageToBrowserController = new WeakMap(); + + /** @internal */ fingerprintInjector?: FingerprintInjector; + fingerprintGenerator?: FingerprintGenerator; + + /** @internal */ fingerprintCache?: QuickLRU; + readonly #maxOpenPagesPerBrowser: number; + readonly #retireBrowserAfterPageCount: number; + readonly #operationTimeoutMillis: number; + readonly #closeInactiveBrowserAfterMillis: number; + + readonly #preLaunchHooks: PreLaunchHook[]; + readonly #postLaunchHooks: PostLaunchHook[]; + readonly #prePageCreateHooks: PrePageCreateHook[]; + readonly #postPageCreateHooks: PostPageCreateHook[]; + readonly #prePageCloseHooks: PrePageCloseHook[]; + readonly #postPageCloseHooks: PostPageCloseHook[]; + + #pageCounter = 0; + // kept as TS-private rather than `#`: the page-close tests observe page tracking and + // controller retirement directly. Excluded from the public surface map either way. + private pages = new Map(); + #pageIds = new WeakMap(); + #startingBrowserControllers = new Set(); + private activeBrowserControllers = new Set(); + private retiredBrowserControllers = new Set(); + #pageToBrowserController = new WeakMap(); + // kept as TS-private: tests replace this interval through bracket access private browserKillerInterval? = setInterval( async () => this.closeInactiveRetiredBrowsers(), @@ -401,13 +414,12 @@ export class BrowserPool< } this.browserPlugins = browserPlugins as unknown as BrowserPlugins; - this.maxOpenPagesPerBrowser = maxOpenPagesPerBrowser; this.maxOpenBrowsers = Infinity; - this.retireBrowserAfterPageCount = retireBrowserAfterPageCount; - this.operationTimeoutMillis = operationTimeoutSecs * 1000; - this.closeInactiveBrowserAfterMillis = closeInactiveBrowserAfterSecs * 1000; - this.useFingerprints = useFingerprints; this.fingerprintOptions = fingerprintOptions; + this.#maxOpenPagesPerBrowser = maxOpenPagesPerBrowser; + this.#retireBrowserAfterPageCount = retireBrowserAfterPageCount; + this.#operationTimeoutMillis = operationTimeoutSecs * 1000; + this.#closeInactiveBrowserAfterMillis = closeInactiveBrowserAfterSecs * 1000; this.#browserRetireInterval = setInterval( async () => @@ -425,16 +437,25 @@ export class BrowserPool< this.#browserRetireInterval!.unref(); // hooks - this.preLaunchHooks = preLaunchHooks; - this.postLaunchHooks = postLaunchHooks; - this.prePageCreateHooks = prePageCreateHooks; - this.postPageCreateHooks = postPageCreateHooks; - this.prePageCloseHooks = prePageCloseHooks; - this.postPageCloseHooks = postPageCloseHooks; + this.#preLaunchHooks = preLaunchHooks; + this.#postLaunchHooks = postLaunchHooks; + this.#prePageCreateHooks = prePageCreateHooks; + this.#postPageCreateHooks = postPageCreateHooks; + this.#prePageCloseHooks = prePageCloseHooks; + this.#postPageCloseHooks = postPageCloseHooks; // fingerprinting - if (this.useFingerprints) { + if (useFingerprints) { this.initializeFingerprinting(); + + // The fingerprint pre-launch hook goes last because of the fingerprint cache. + // It is usual to generate proxy per browser and we want to know the proxyUrl for the caching. + this.#preLaunchHooks = [...this.#preLaunchHooks, createFingerprintPreLaunchHook(this)]; + this.#prePageCreateHooks = [createPrePageCreateHook(), ...this.#prePageCreateHooks]; + this.#postPageCreateHooks = [ + createPostPageCreateHook(this.fingerprintInjector!), + ...this.#postPageCreateHooks, + ]; } } @@ -567,7 +588,7 @@ export class BrowserPool< * @param page - Browser plugin page */ getBrowserControllerByPage(page: PageReturn): BrowserControllerReturn | undefined { - return this.pageToBrowserController.get(page); + return this.#pageToBrowserController.get(page); } /** @@ -587,7 +608,7 @@ export class BrowserPool< * until it's closed. */ getPageId(page: PageReturn): string | undefined { - return this.pageIds.get(page); + return this.#pageIds.get(page); } private async createPageForBrowser( @@ -615,7 +636,7 @@ export class BrowserPool< } } - await this.executeHooks(this.prePageCreateHooks, pageId, browserController, finalPageOptions); + await this.executeHooks(this.#prePageCreateHooks, pageId, browserController, finalPageOptions); tryCancel(); let page: PageReturn; @@ -623,17 +644,17 @@ export class BrowserPool< try { page = (await addTimeoutToPromise( async () => browserController.newPage(finalPageOptions), - this.operationTimeoutMillis, + this.#operationTimeoutMillis, 'browserController.newPage() timed out.', )) as PageReturn; tryCancel(); this.pages.set(pageId, page); - this.pageIds.set(page, pageId); - this.pageToBrowserController.set(page, browserController); + this.#pageIds.set(page, pageId); + this.#pageToBrowserController.set(page, browserController); // if you synchronously trigger a lot of page launches, browser will not get retired soon enough. Not sure if it's a problem, let's monitor it. - if (browserController.totalPages >= this.retireBrowserAfterPageCount) { + if (browserController.totalPages >= this.#retireBrowserAfterPageCount) { this.retireBrowserController(browserController); } @@ -645,7 +666,7 @@ export class BrowserPool< ); } - await this.executeHooks(this.postPageCreateHooks, page, browserController); + await this.executeHooks(this.#postPageCreateHooks, page, browserController); tryCancel(); this.emit(BROWSER_POOL_EVENTS.PAGE_CREATED, page); @@ -659,7 +680,7 @@ export class BrowserPool< * */ retireBrowserController(browserController: BrowserControllerReturn): void { - const isStarting = this.startingBrowserControllers.has(browserController); + const isStarting = this.#startingBrowserControllers.has(browserController); const isActive = this.activeBrowserControllers.has(browserController); const hasBeenRetiredOrKilled = !isStarting && !isActive; @@ -667,7 +688,7 @@ export class BrowserPool< this.retiredBrowserControllers.add(browserController); this.emit(BROWSER_POOL_EVENTS.BROWSER_RETIRED, browserController); - this.startingBrowserControllers.delete(browserController); + this.#startingBrowserControllers.delete(browserController); this.activeBrowserControllers.delete(browserController); } @@ -748,7 +769,7 @@ export class BrowserPool< * closed after all their pages are closed. */ retireAllBrowsers(): void { - [...this.startingBrowserControllers, ...this.activeBrowserControllers].forEach((controller) => { + [...this.#startingBrowserControllers, ...this.activeBrowserControllers].forEach((controller) => { this.retireBrowserController(controller); }); } @@ -777,7 +798,7 @@ export class BrowserPool< async releaseAllBrowsers(): Promise { await this.closeAllBrowsers(); - this.startingBrowserControllers.clear(); + this.#startingBrowserControllers.clear(); this.activeBrowserControllers.clear(); this.retiredBrowserControllers.clear(); } @@ -799,7 +820,7 @@ export class BrowserPool< private getAllBrowserControllers() { return new Set([ - ...this.startingBrowserControllers, + ...this.#startingBrowserControllers, ...this.activeBrowserControllers, ...this.retiredBrowserControllers, ]); @@ -809,7 +830,7 @@ export class BrowserPool< const { browserPlugin, launchOptions, proxyUrl, ignoreTlsErrors } = options; const browserController = browserPlugin.createController() as BrowserControllerReturn; - this.startingBrowserControllers.add(browserController); + this.#startingBrowserControllers.add(browserController); const launchContext = browserPlugin.createLaunchContext({ id: pageId, @@ -830,13 +851,13 @@ export class BrowserPool< try { // If the hooks or the launch fails, we need to delete the controller, // because otherwise it would be stuck in limbo without a browser. - await this.executeHooks(this.preLaunchHooks, pageId, launchContext); + await this.executeHooks(this.#preLaunchHooks, pageId, launchContext); tryCancel(); const browser = await browserPlugin.launch(launchContext); tryCancel(); browserController.assignBrowser(browser, launchContext); } catch (err) { - this.startingBrowserControllers.delete(browserController); + this.#startingBrowserControllers.delete(browserController); throw err; } @@ -846,9 +867,9 @@ export class BrowserPool< try { // If the launch fails on the post-launch hooks, we need to clean up // both the controller and the browser before throwing. - await this.executeHooks(this.postLaunchHooks, pageId, browserController); + await this.executeHooks(this.#postLaunchHooks, pageId, browserController); } catch (err) { - this.startingBrowserControllers.delete(browserController); + this.#startingBrowserControllers.delete(browserController); browserController.close().catch((closeErr) => { this.#log.error(`Could not close browser whose post-launch hooks failed.\nCause:${closeErr.message}`, { id: browserController.id, @@ -859,7 +880,7 @@ export class BrowserPool< tryCancel(); browserController.activate(); - this.startingBrowserControllers.delete(browserController); + this.#startingBrowserControllers.delete(browserController); this.activeBrowserControllers.add(browserController); this.emit(BROWSER_POOL_EVENTS.BROWSER_LAUNCHED, browserController); @@ -871,15 +892,15 @@ export class BrowserPool< * @private */ private pickBrowserPlugin() { - const pluginIndex = this.pageCounter % this.browserPlugins.length; - this.pageCounter++; + const pluginIndex = this.#pageCounter % this.browserPlugins.length; + this.#pageCounter++; return this.browserPlugins[pluginIndex]; } private pickBrowserWithFreeCapacity(browserPlugin: BrowserPlugin, options?: { proxyUrl?: string }) { return [...this.activeBrowserControllers].find((controller) => { - const hasCapacity = controller.activePages < this.maxOpenPagesPerBrowser; + const hasCapacity = controller.activePages < this.#maxOpenPagesPerBrowser; const isCorrectPlugin = controller.browserPlugin === browserPlugin; const isSameProxyUrl = controller.proxyUrl === options?.proxyUrl; @@ -898,7 +919,7 @@ export class BrowserPool< for (const controller of this.retiredBrowserControllers) { const millisSinceLastPageOpened = Date.now() - controller.lastPageOpenedAt; - const isBrowserIdle = millisSinceLastPageOpened >= this.closeInactiveBrowserAfterMillis; + const isBrowserIdle = millisSinceLastPageOpened >= this.#closeInactiveBrowserAfterMillis; const isBrowserEmpty = controller.activePages === 0; if (isBrowserIdle || isBrowserEmpty) { @@ -920,7 +941,7 @@ export class BrowserPool< private overridePageClose(page: PageReturn) { const originalPageClose = page.close; - const browserController = this.pageToBrowserController.get(page)!; + const browserController = this.#pageToBrowserController.get(page)!; const pageId = this.getPageId(page)!; page.close = async (...args: unknown[]) => { @@ -934,7 +955,7 @@ export class BrowserPool< let pageClosed = false; const closing = (async () => { - await this.executeHooks(this.prePageCloseHooks, page, browserController); + await this.executeHooks(this.#prePageCloseHooks, page, browserController); await originalPageClose.apply(page, args).catch((err: Error) => { this.#log.debug(`Could not close page.\nCause:${err.message}`, { id: browserController.id }); @@ -944,7 +965,7 @@ export class BrowserPool< // so that a slow hook does not get the browser retired. pageClosed = true; - await this.executeHooks(this.postPageCloseHooks, pageId, browserController); + await this.executeHooks(this.#postPageCloseHooks, pageId, browserController); })(); let timeout: NodeJS.Timeout | undefined; @@ -1010,10 +1031,12 @@ export class BrowserPool< /** * Returns `true` if the pool can accept a new browser launch without exceeding * {@link BrowserPoolOptions.maxOpenBrowsers}. Counts starting, active, and retired browsers. + * + * @internal */ hasFreeBrowserSlot(): boolean { const total = - this.startingBrowserControllers.size + + this.#startingBrowserControllers.size + this.activeBrowserControllers.size + this.retiredBrowserControllers.size; return total < this.maxOpenBrowsers; @@ -1021,10 +1044,12 @@ export class BrowserPool< /** * Returns `true` if any active browser has room for another page. + * + * @internal */ hasActiveBrowserWithFreeCapacity(): boolean { for (const controller of this.activeBrowserControllers) { - if (controller.activePages < this.maxOpenPagesPerBrowser) return true; + if (controller.activePages < this.#maxOpenPagesPerBrowser) return true; } return false; } @@ -1037,19 +1062,6 @@ export class BrowserPool< if (useFingerprintCache) { this.fingerprintCache = new QuickLRU({ maxSize: fingerprintCacheSize }); } - - this.addFingerprintHooks(); - } - - private addFingerprintHooks() { - this.preLaunchHooks = [ - ...this.preLaunchHooks, - // This is flipped because of the fingerprint cache. - // It is usual to generate proxy per browser and we want to know the proxyUrl for the caching. - createFingerprintPreLaunchHook(this), - ]; - this.prePageCreateHooks = [createPrePageCreateHook(), ...this.prePageCreateHooks]; - this.postPageCreateHooks = [createPostPageCreateHook(this.fingerprintInjector!), ...this.postPageCreateHooks]; } } diff --git a/packages/browser-pool/src/container-proxy-server.ts b/packages/browser-pool/src/container-proxy-server.ts deleted file mode 100644 index fffc460e2c75..000000000000 --- a/packages/browser-pool/src/container-proxy-server.ts +++ /dev/null @@ -1,51 +0,0 @@ -import { Server as ProxyChainServer } from 'proxy-chain'; - -/** - * Creates a proxy server designed to handle requests from "container" instances. - * Each container instance is assigned to a different (but still localhost) IP address - * in order to work around authorization and to enable upstream. - * @internal - */ -export async function createProxyServerForContainers(fallbackProxyUrl?: string) { - const ipToProxy = new Map(); - - const proxyServer = new ProxyChainServer({ - prepareRequestFunction({ request }) { - const prefix4to6 = '::ffff:'; - const localAddress = request.socket.localAddress!.startsWith(prefix4to6) - ? request.socket.localAddress!.slice(prefix4to6.length) - : request.socket.localAddress!; - - const upstreamProxyUrl = ipToProxy.get(localAddress); - - if (upstreamProxyUrl === undefined) { - if (fallbackProxyUrl) { - return { - upstreamProxyUrl: fallbackProxyUrl, - requestAuthentication: false, - }; - } - - console.warn(`Request without proxy ${localAddress} ${request.headers.host}`); - } - - return { - upstreamProxyUrl, - requestAuthentication: false, - }; - }, - port: 0, - }); - - await proxyServer.listen(); - - proxyServer.server.unref(); - - return { - port: proxyServer.port, - ipToProxy, - async close(closeConnections: boolean) { - return proxyServer.close(closeConnections); - }, - }; -} diff --git a/packages/browser-pool/src/events.ts b/packages/browser-pool/src/events.ts index ab07c91a52a1..14b6102bff96 100644 --- a/packages/browser-pool/src/events.ts +++ b/packages/browser-pool/src/events.ts @@ -1,7 +1,6 @@ export enum BROWSER_POOL_EVENTS { BROWSER_LAUNCHED = 'browserLaunched', BROWSER_RETIRED = 'browserRetired', - BROWSER_CLOSED = 'browserClosed', PAGE_CREATED = 'pageCreated', PAGE_CLOSED = 'pageClosed', diff --git a/packages/browser-pool/src/fingerprinting/types.ts b/packages/browser-pool/src/fingerprinting/types.ts index fdb8c5b40953..bb15b24af108 100644 --- a/packages/browser-pool/src/fingerprinting/types.ts +++ b/packages/browser-pool/src/fingerprinting/types.ts @@ -1,25 +1,7 @@ -import type { - BrowserFingerprintWithHeaders as Fingerprint, - FingerprintGeneratorOptions as FingerprintOptionsOriginal, -} from 'fingerprint-generator'; - -export interface FingerprintGenerator { - getFingerprint: (fingerprintGeneratorOptions?: FingerprintGeneratorOptions) => GetFingerprintReturn; -} - -export interface GetFingerprintReturn { - fingerprint: Fingerprint; -} +import type { FingerprintGeneratorOptions as FingerprintOptionsOriginal } from 'fingerprint-generator'; export interface FingerprintGeneratorOptions extends Partial {} -const SUPPORTED_HTTP_VERSIONS = ['1', '2'] as const; - -/** - * String specifying the HTTP version to use. - */ -type HttpVersion = (typeof SUPPORTED_HTTP_VERSIONS)[number]; - export enum BrowserName { chrome = 'chrome', firefox = 'firefox', @@ -27,25 +9,6 @@ export enum BrowserName { edge = 'edge', } -export interface BrowserSpecification { - /** - * String representing the browser name. - */ - name: BrowserName; - /** - * Minimum version of browser used. - */ - minVersion?: number; - /** - * Maximum version of browser used. - */ - maxVersion?: number; - /** - * HTTP version to be used for header generation (the headers differ depending on the version). - */ - httpVersion?: HttpVersion; -} - export enum OperatingSystemsName { linux = 'linux', macos = 'macos', diff --git a/packages/browser-pool/src/index.ts b/packages/browser-pool/src/index.ts index 9f3b49455248..33638b08851e 100644 --- a/packages/browser-pool/src/index.ts +++ b/packages/browser-pool/src/index.ts @@ -26,12 +26,7 @@ export * from './browser-pool.js'; export * from './playwright/playwright-plugin.js'; export * from './puppeteer/puppeteer-plugin.js'; export * from './events.js'; -export type { - BrowserSpecification, - FingerprintGenerator, - FingerprintGeneratorOptions, - GetFingerprintReturn, -} from './fingerprinting/types.js'; +export type { FingerprintGeneratorOptions } from './fingerprinting/types.js'; export { BrowserName, DeviceCategory, OperatingSystemsName } from './fingerprinting/types.js'; export type { BrowserControllerEvents, diff --git a/packages/browser-pool/src/launch-context.ts b/packages/browser-pool/src/launch-context.ts index 14f02e18045d..858b3fa6cd72 100644 --- a/packages/browser-pool/src/launch-context.ts +++ b/packages/browser-pool/src/launch-context.ts @@ -60,6 +60,7 @@ export interface LaunchContextOptions< * Whether this launch context represents a connection to a remote browser * rather than a locally launched one. * @default false + * @internal */ isRemote?: boolean; } diff --git a/packages/browser-pool/src/playwright/playwright-browser.ts b/packages/browser-pool/src/playwright/playwright-browser.ts index 20cede4bd27a..b9d649a38288 100644 --- a/packages/browser-pool/src/playwright/playwright-browser.ts +++ b/packages/browser-pool/src/playwright/playwright-browser.ts @@ -16,6 +16,7 @@ export class PlaywrightBrowser extends EventEmitter { #isConnected = true; #browserType?: BrowserType; + /** @internal */ constructor(options: BrowserOptions) { super(); diff --git a/packages/browser-pool/src/playwright/playwright-controller.ts b/packages/browser-pool/src/playwright/playwright-controller.ts index 88ea3b8ccf46..68f3af368fe5 100644 --- a/packages/browser-pool/src/playwright/playwright-controller.ts +++ b/packages/browser-pool/src/playwright/playwright-controller.ts @@ -12,6 +12,7 @@ export class PlaywrightController extends BrowserController< SafeParameters[0], Browser > { + /** @internal */ normalizeProxyOptions(proxyUrl: string | undefined, pageOptions: any): Record { if (!proxyUrl) { return {}; diff --git a/packages/browser-pool/src/playwright/playwright-plugin.ts b/packages/browser-pool/src/playwright/playwright-plugin.ts index bcc24a0619fb..59c04ec60c95 100644 --- a/packages/browser-pool/src/playwright/playwright-plugin.ts +++ b/packages/browser-pool/src/playwright/playwright-plugin.ts @@ -21,6 +21,8 @@ export class PlaywrightPlugin extends BrowserPlugin< /** * Playwright remote connections only support incognito pages — `connect()` / `connectOverCDP()` don't * accept persistent contexts. Force it on (and inform the user) when wired for a remote connection. + * + * @internal */ override useRemoteConnection(connection: RemoteConnection, parameters: RemoteConnectionParameters = {}): void { super.useRemoteConnection(connection, parameters); diff --git a/packages/browser-pool/src/puppeteer/puppeteer-controller.ts b/packages/browser-pool/src/puppeteer/puppeteer-controller.ts index f9367314f395..c912b879ec02 100644 --- a/packages/browser-pool/src/puppeteer/puppeteer-controller.ts +++ b/packages/browser-pool/src/puppeteer/puppeteer-controller.ts @@ -22,6 +22,7 @@ export class PuppeteerController extends BrowserController< PuppeteerTypes.Browser, PuppeteerNewPageOptions > { + /** @internal */ normalizeProxyOptions(proxyUrl: string | undefined, pageOptions: any): Record { if (!proxyUrl) { return {}; diff --git a/packages/browser-pool/src/puppeteer/puppeteer-plugin.ts b/packages/browser-pool/src/puppeteer/puppeteer-plugin.ts index 1f8c29339ae6..620780dc4a64 100644 --- a/packages/browser-pool/src/puppeteer/puppeteer-plugin.ts +++ b/packages/browser-pool/src/puppeteer/puppeteer-plugin.ts @@ -22,7 +22,11 @@ export class PuppeteerPlugin extends BrowserPlugin< PuppeteerTypes.Browser, PuppeteerNewPageOptions > { - /** Pages share cookies/storage on the remote browser (Puppeteer defaults to non-incognito). */ + /** + * Pages share cookies/storage on the remote browser (Puppeteer defaults to non-incognito). + * + * @internal + */ override useRemoteConnection(connection: RemoteConnection, parameters: RemoteConnectionParameters = {}): void { super.useRemoteConnection(connection, parameters); diff --git a/packages/browser-pool/src/remote-browser-pool.ts b/packages/browser-pool/src/remote-browser-pool.ts index 00e81115b241..c7e30c6482a9 100644 --- a/packages/browser-pool/src/remote-browser-pool.ts +++ b/packages/browser-pool/src/remote-browser-pool.ts @@ -154,7 +154,11 @@ export interface RemoteBrowserPoolOptions { connection?: RemoteConnectionParameters; /** Extra {@apilink BrowserPool} options (lifecycle hooks, page limits, fingerprinting, …). */ browserPoolOptions?: Omit & BrowserPoolHooks; - /** Fallback poll interval (ms) while waiting for a free browser slot. The wait is event-driven; this only bounds it. @default 500 */ + /** + * Fallback poll interval (ms) while waiting for a free browser slot. The wait is event-driven; this only bounds it. + * @default 500 + * @internal + */ slotPollIntervalMillis?: number; } @@ -198,7 +202,11 @@ export type CrawlerRemoteBrowserOptions = Omit implements IBrowserPool { - /** The wrapped pool that performs the remote connections and serves pages. */ + /** + * The wrapped pool that performs the remote connections and serves pages. + * + * @internal + */ readonly browserPool: BrowserPool; /** The wrapped pool viewed through the {@apilink IBrowserPool} contract (the bare type widens pages to `never`). */ diff --git a/packages/browser-pool/test/multiple-plugins.test.ts b/packages/browser-pool/test/multiple-plugins.test.ts index 444be6a6cf08..e2dc51daea35 100644 --- a/packages/browser-pool/test/multiple-plugins.test.ts +++ b/packages/browser-pool/test/multiple-plugins.test.ts @@ -1,4 +1,4 @@ -import { BrowserPool, PlaywrightPlugin } from '@crawlee/browser-pool'; +import { BROWSER_POOL_EVENTS, BrowserPool, PlaywrightPlugin } from '@crawlee/browser-pool'; import playwright from 'playwright'; describe('BrowserPool - Using multiple plugins', () => { @@ -9,14 +9,21 @@ describe('BrowserPool - Using multiple plugins', () => { }>; const chromePlugin = new PlaywrightPlugin(playwright.chromium); const firefoxPlugin = new PlaywrightPlugin(playwright.firefox); + // The pool's browser bookkeeping is private, so we count launches through the public event. + // No browser is retired in these tests, so this is also the number of active browsers. + let launchedBrowsers = 0; beforeEach(async () => { vitest.clearAllMocks(); + launchedBrowsers = 0; browserPool = new BrowserPool({ browserPlugins: [chromePlugin, firefoxPlugin], closeInactiveBrowserAfterSecs: 2, retireInactiveBrowserAfterSecs: 30, }); + browserPool.on(BROWSER_POOL_EVENTS.BROWSER_LAUNCHED, () => { + launchedBrowsers++; + }); }); afterEach(async () => { @@ -40,7 +47,7 @@ describe('BrowserPool - Using multiple plugins', () => { const pages = await Promise.all(pagePromises); expect(pages).toHaveLength(correctPluginOrder.length); - expect(browserPool.activeBrowserControllers.size).toEqual(2); + expect(launchedBrowsers).toEqual(2); for (const [idx, page] of pages.entries()) { const controller = browserPool.getBrowserControllerByPage(page)!; @@ -67,12 +74,12 @@ describe('BrowserPool - Using multiple plugins', () => { await browserPool.newPage(); expect(chromePlugin.launch).toHaveBeenCalledTimes(1); expect(firefoxPlugin.launch).toHaveBeenCalledTimes(1); - expect(browserPool.activeBrowserControllers.size).toBe(2); + expect(launchedBrowsers).toBe(2); // Open more pages await browserPool.newPageWithEachPlugin(); expect(chromePlugin.launch).toHaveBeenCalledTimes(1); expect(firefoxPlugin.launch).toHaveBeenCalledTimes(1); - expect(browserPool.activeBrowserControllers.size).toBe(2); + expect(launchedBrowsers).toBe(2); }); }); diff --git a/packages/cheerio-crawler/src/internals/cheerio-crawler.ts b/packages/cheerio-crawler/src/internals/cheerio-crawler.ts index 16d81fcac6f8..0b642dde2b39 100644 --- a/packages/cheerio-crawler/src/internals/cheerio-crawler.ts +++ b/packages/cheerio-crawler/src/internals/cheerio-crawler.ts @@ -38,11 +38,6 @@ export interface CheerioCrawlerOptions< StatisticStateExtension > {} -export type CheerioHook< - UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler - JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler -> = InternalHttpHook>; - export interface CheerioCrawlingContext< UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler diff --git a/packages/core/src/memory-storage/memory-storage.ts b/packages/core/src/memory-storage/memory-storage.ts index d946ddfece57..047c341c0c20 100644 --- a/packages/core/src/memory-storage/memory-storage.ts +++ b/packages/core/src/memory-storage/memory-storage.ts @@ -139,7 +139,7 @@ export class MemoryStorageBackend implements storage.StorageBackend { return newStore; } - async createRequestQueueBackend(options: storage.StorageIdentifier = {}): Promise { + async createRequestQueueBackend(options: storage.StorageIdentifier = {}): Promise { const { isAlias, cacheKey } = MemoryStorageBackend.#resolveStorageKey(options); const found = this.#requestQueueBackendCache.find( diff --git a/packages/core/src/proxy_configuration.ts b/packages/core/src/proxy_configuration.ts index fd1c2ac7cfe0..81e15c69b1cd 100644 --- a/packages/core/src/proxy_configuration.ts +++ b/packages/core/src/proxy_configuration.ts @@ -16,15 +16,13 @@ export interface ProxyConfigurationFunction { (options?: { request?: Request }): string | null | Promise; } -type UrlList = (string | null)[]; - export interface ProxyConfigurationOptions { /** * An array of custom proxy URLs to be rotated. * Custom proxies are not compatible with Apify Proxy and an attempt to use both * configuration options will cause an error to be thrown on initialize. */ - proxyUrls?: UrlList; + proxyUrls?: (string | null)[]; /** * Custom function that allows you to generate the new proxy URL dynamically. It gets an optional parameter with the `Request` object when applicable. @@ -33,10 +31,14 @@ export interface ProxyConfigurationOptions { * This function is used to generate the URL when {@apilink ProxyConfiguration.newUrl} or {@apilink ProxyConfiguration.newProxyInfo} is called. */ newUrlFunction?: ProxyConfigurationFunction; -} -interface NewUrlOptions { - request?: Request; + /** + * When truthy, the constructor throws unless one of `proxyUrls` / `newUrlFunction` was given. Falsy by + * default, so a bare `ProxyConfiguration` can be constructed. Set by the Apify SDK, which builds the options + * object itself; declared here only so that it stays type-checkable. + * @internal + */ + validateRequired?: boolean; } /** @@ -54,7 +56,7 @@ export interface IProxyConfiguration { * Creates a new {@apilink ProxyInfo} object describing the proxy to use for the given * request. Returns `undefined` when no proxy should be used. */ - newProxyInfo(options?: NewUrlOptions): Promise; + newProxyInfo(options?: { request?: Request }): Promise; } /** @@ -88,7 +90,7 @@ export interface IProxyConfiguration { export class ProxyConfiguration implements IProxyConfiguration { readonly isManInTheMiddle = false; #nextCustomUrlIndex = 0; - #proxyUrls?: UrlList; + #proxyUrls?: (string | null)[]; #newUrlFunction?: ProxyConfigurationFunction; /** @@ -112,6 +114,9 @@ export class ProxyConfiguration implements IProxyConfiguration { * ``` */ constructor(options: ProxyConfigurationOptions = {}) { + // `validateRequired` is destructured off before the strict-object parse on purpose: the Apify SDK passes it + // through a computed key (`['validateRequired' as string]: false`), and leaving it in `rest` would make + // `Actor.createProxyConfiguration()` fail the `z.strictObject` check with a `ZodError`. const { validateRequired, ...rest } = options as Dictionary; if ('tieredProxyUrls' in rest) { @@ -142,7 +147,7 @@ export class ProxyConfiguration implements IProxyConfiguration { * * @return Represents information about used proxy and its configuration. */ - async newProxyInfo(options?: NewUrlOptions): Promise { + async newProxyInfo(options?: { request?: Request }): Promise { const url = await this.newUrl(options); if (!url) return undefined; @@ -163,7 +168,7 @@ export class ProxyConfiguration implements IProxyConfiguration { * @return A string with a proxy URL, including authentication credentials and port number. * For example, `http://bob:password123@proxy.example.com:8000` */ - async newUrl(options?: NewUrlOptions): Promise { + async newUrl(options?: { request?: Request }): Promise { if (this.#newUrlFunction) { return (await this.callNewUrlFunction({ request: options?.request })) ?? undefined; } diff --git a/packages/core/src/request.ts b/packages/core/src/request.ts index af53e08adfaa..432a28f4ef41 100644 --- a/packages/core/src/request.ts +++ b/packages/core/src/request.ts @@ -412,6 +412,7 @@ class CrawleeRequest { /** * Reason for skipping this request. + * @internal */ get skippedReason(): SkippedRequestReason | undefined { return this.userData.__crawlee?.skippedReason; @@ -419,6 +420,7 @@ class CrawleeRequest { /** * Reason for skipping this request. + * @internal */ set skippedReason(value: SkippedRequestReason | undefined) { if (!this.userData.__crawlee) { diff --git a/packages/core/src/service_locator.ts b/packages/core/src/service_locator.ts index d8eba146dd32..22ca448eaa02 100644 --- a/packages/core/src/service_locator.ts +++ b/packages/core/src/service_locator.ts @@ -80,6 +80,7 @@ interface ServiceLocatorInterface { /** * Get the storage instance manager (shared across all storage types). + * @internal */ getStorageInstanceManager(): StorageInstanceManager; @@ -295,6 +296,7 @@ export class ServiceLocator implements ServiceLocatorInterface { return this.getLogger().child({ prefix }); } + /** @internal */ getStorageInstanceManager(): StorageInstanceManager { if (!ServiceLocator.#storageInstanceManager) { ServiceLocator.#storageInstanceManager = new StorageInstanceManager(); @@ -302,6 +304,7 @@ export class ServiceLocator implements ServiceLocatorInterface { return ServiceLocator.#storageInstanceManager; } + /** @internal */ reset(): void { this.#configuration = undefined; this.#eventManager = undefined; diff --git a/packages/core/src/storages/dataset.ts b/packages/core/src/storages/dataset.ts index c8b35973b65b..82641e34b57d 100644 --- a/packages/core/src/storages/dataset.ts +++ b/packages/core/src/storages/dataset.ts @@ -4,7 +4,6 @@ import { z } from 'zod'; import { tryCancel } from '@apify/timeout'; import { Configuration } from '../configuration.js'; -import type { CrawleeLogger } from '../log.js'; import { serviceLocator } from '../service_locator.js'; import { parseArgument, schemas, validators } from '../validators.js'; import type { DatasetJournalEntry, JournalEntry } from './transaction.js'; @@ -192,10 +191,10 @@ export interface DatasetExportToOptions extends DatasetExportOptions { * @category Result Stores */ export class Dataset { - id: string; - name?: string; - backend: DatasetBackend; - log: CrawleeLogger; + readonly id: string; + readonly name?: string; + // kept as TS-private: dataset tests spy on the backend directly + private readonly backend: DatasetBackend; readonly #statsTracker = new StorageStatsTracker({ readCount: 0, @@ -212,7 +211,6 @@ export class Dataset { this.id = options.metadata.id; this.name = options.metadata.name; this.backend = options.backend; - this.log = serviceLocator.getLogger().child({ prefix: 'Dataset' }); } /** @@ -905,6 +903,7 @@ export interface DatasetReducer { (memo: T, item: Data, index: number): Awaitable; } +/** @internal */ export interface DatasetOptions { /** Resolved metadata for the dataset, as returned by the backend's `getMetadata()`. */ metadata: DatasetInfo; diff --git a/packages/core/src/storages/index.ts b/packages/core/src/storages/index.ts index e520a550b087..05dcca6e2cb8 100644 --- a/packages/core/src/storages/index.ts +++ b/packages/core/src/storages/index.ts @@ -5,8 +5,16 @@ export * from './request_list.js'; export type * from './request_loader.js'; export type * from './request_manager.js'; export * from './request_queue.js'; -export * from './storage_instance_manager.js'; -export * from './storage_stats.js'; +// `resolveStorageIdentifier` is deliberately absent: it is an internal helper of the storage frontends. +export type { + DefaultStorageIdentifier, + ExplicitStorageIdentifier, + IStorage, + StorageIdentifier, +} from './storage_instance_manager.js'; +export { StorageInstanceManager } from './storage_instance_manager.js'; +// `StorageStatsTracker` is deliberately absent: it is the mutable counter backing the `stats` getters. +export type { DatasetStats, KeyValueStoreStats, RequestQueueStats } from './storage_stats.js'; export * from './utils.js'; export * from './transaction.js'; export * from './request_manager_tandem.js'; diff --git a/packages/core/src/storages/key_value_store.ts b/packages/core/src/storages/key_value_store.ts index f4f228ffa960..19de9c38cab4 100644 --- a/packages/core/src/storages/key_value_store.ts +++ b/packages/core/src/storages/key_value_store.ts @@ -1060,6 +1060,7 @@ export interface KeyConsumer { (key: string, index: number, info: { size: number }): Awaitable; } +/** @internal */ export interface KeyValueStoreOptions { /** Resolved metadata for the key-value store, as returned by the backend's `getMetadata()`. */ metadata: KeyValueStoreInfo; diff --git a/packages/core/src/storages/request_list.ts b/packages/core/src/storages/request_list.ts index fa8affde6de5..fdbdce217d82 100644 --- a/packages/core/src/storages/request_list.ts +++ b/packages/core/src/storages/request_list.ts @@ -161,7 +161,7 @@ export interface RequestListOptions { * ] * ``` */ - sources?: RequestListSource[]; + sources?: (string | Source)[]; /** * A function that will be called to get the sources for the `RequestList`, but only if `RequestList` @@ -391,7 +391,7 @@ export class RequestList implements IRequestLoader { #persistRequestsKey?: string; #store?: KeyValueStore; #keepDuplicateUrls: boolean; - #sources: RequestListSource[]; + #sources: (string | Source)[]; #sourcesFunction?: RequestListSourcesFunction; #proxyConfiguration?: IProxyConfiguration; #httpClient?: BaseHttpClient; @@ -742,7 +742,7 @@ export class RequestList implements IRequestLoader { * If the `source` parameter is a string or plain object and not an instance * of a `Request`, then the function creates a `Request` instance. */ - private addRequest(source: RequestListSource) { + private addRequest(source: string | Source) { let request: Request | RequestOptions; const type = typeof source; @@ -903,7 +903,7 @@ export class RequestList implements IRequestLoader { */ static async open( listNameOrOptions: string | null | RequestListOptions, - sources?: RequestListSource[], + sources?: (string | Source)[], options: RequestListOptions = {}, ): Promise { if (listNameOrOptions != null && typeof listNameOrOptions === 'object') { @@ -973,5 +973,4 @@ export interface RequestListState { inProgress: string[]; } -type RequestListSource = string | Source; -export type RequestListSourcesFunction = () => Promise; +export type RequestListSourcesFunction = () => Promise<(string | Source)[]>; diff --git a/packages/core/src/storages/request_queue.ts b/packages/core/src/storages/request_queue.ts index 0ce947116f87..75cb0695e874 100644 --- a/packages/core/src/storages/request_queue.ts +++ b/packages/core/src/storages/request_queue.ts @@ -1089,6 +1089,7 @@ interface RequestLruItem { forefront: boolean; } +/** @internal */ export interface RequestQueueOptions { /** Resolved metadata for the request queue, as returned by the backend's `getMetadata()`. */ metadata: RequestQueueInfo; diff --git a/packages/core/src/storages/storage_instance_manager.ts b/packages/core/src/storages/storage_instance_manager.ts index f1ec38d9ee57..2105b62e0636 100644 --- a/packages/core/src/storages/storage_instance_manager.ts +++ b/packages/core/src/storages/storage_instance_manager.ts @@ -18,12 +18,10 @@ export interface IStorage { name?: string; } -type Hashable = string; - /** Reserved alias for the default (unnamed) storage. */ const DEFAULT_STORAGE_ALIAS = '__default__'; -type CacheTier = Map, Map>>; +type CacheTier = Map, Map>>; /** * Three-tier cache for storage instances, modelled after crawlee-python's `_StorageCache`. @@ -49,7 +47,7 @@ class StorageCache { | { id: string; name?: string; alias?: undefined } | { id?: string; name: string; alias?: undefined } | { id?: undefined; name?: undefined; alias: string } - ) & { backendCacheKey: Hashable }, + ) & { backendCacheKey: string }, ): T | undefined { for (const [tier, key] of [ [this.byId, id], @@ -75,7 +73,7 @@ class StorageCache { cls: Constructor, key: string, instance: T, - backendCacheKey: Hashable, + backendCacheKey: string, ): void { if (!tier.has(cls)) tier.set(cls, new Map()); const keyMap = tier.get(cls)!; @@ -86,7 +84,7 @@ class StorageCache { /** * Cache an instance under its actual id, name, and an optional alias. */ - set(cls: Constructor, instance: T, backendCacheKey: Hashable, alias?: string): void { + set(cls: Constructor, instance: T, backendCacheKey: string, alias?: string): void { // Always cache by id. this.setInMap(this.byId, cls, instance.id, instance, backendCacheKey); @@ -124,7 +122,7 @@ class StorageCache { */ checkNameAliasConflict( cls: Constructor, - { name, alias, backendCacheKey }: { name?: string; alias?: string; backendCacheKey: Hashable }, + { name, alias, backendCacheKey }: { name?: string; alias?: string; backendCacheKey: string }, ): void { if (alias) { const existingByName = this.byName.get(cls)?.get(alias)?.get(backendCacheKey); @@ -204,7 +202,7 @@ export class StorageInstanceManager { backendCacheKey, }: (ExplicitStorageIdentifier | DefaultStorageIdentifier) & { backendOpener: () => Promise; - backendCacheKey: Hashable; + backendCacheKey: string; }, ): Promise { // Auto-set alias='__default__' when no parameters are specified (mirrors crawlee-python). diff --git a/packages/core/src/storages/transaction.ts b/packages/core/src/storages/transaction.ts index ca6771c45b15..63d031c1e30c 100644 --- a/packages/core/src/storages/transaction.ts +++ b/packages/core/src/storages/transaction.ts @@ -54,6 +54,7 @@ export interface TransactionParticipant { /** * A single dataset write (`pushData`) recorded in a transaction journal. + * @internal */ export interface DatasetJournalEntry { type: 'dataset'; @@ -67,6 +68,7 @@ export interface DatasetJournalEntry { /** * A single key-value store write (`setValue`) recorded in a transaction journal. + * @internal */ export interface KeyValueStoreJournalEntry { type: 'keyValueStore'; @@ -81,6 +83,7 @@ export interface KeyValueStoreJournalEntry { /** * A request recorded in a transaction journal. + * @internal */ export interface JournaledRequest { url: string; @@ -95,6 +98,7 @@ export interface JournaledRequest { /** * A batch of request queue additions recorded in a transaction journal. + * @internal */ export interface RequestQueueJournalEntry { type: 'requestQueue'; @@ -106,6 +110,7 @@ export interface RequestQueueJournalEntry { writeThrough: boolean; } +/** @internal */ export type JournalEntry = DatasetJournalEntry | KeyValueStoreJournalEntry | RequestQueueJournalEntry; /** @@ -164,11 +169,13 @@ const COMMIT_ORDER: JournalEntry['type'][] = ['keyValueStore', 'requestQueue', ' * handler unless `transactionalStorage: false` is set. */ export class StorageTransaction implements StorageTransactionView { - /** The ordered, append-only journal — the source of truth for commit, introspection and reads. */ + /** + * The ordered, append-only journal — the source of truth for commit, introspection and reads. + * @internal + */ readonly journal: JournalEntry[] = []; - /** Per-storage-type write policy. */ - readonly policy: StorageWritePolicy; + readonly #policy: StorageWritePolicy; readonly #commitTimeoutMillis: number; @@ -181,10 +188,18 @@ export class StorageTransaction implements StorageTransactionView { /** @internal */ constructor(options: StorageTransactionOptions = {}) { - this.policy = { ...DEFAULT_STORAGE_WRITE_POLICY, ...options.policy }; + this.#policy = { ...DEFAULT_STORAGE_WRITE_POLICY, ...options.policy }; this.#commitTimeoutMillis = options.commitTimeoutMillis ?? DEFAULT_COMMIT_TIMEOUT_MILLIS; } + /** + * Per-storage-type write policy. + * @internal + */ + get policy(): StorageWritePolicy { + return this.#policy; + } + get state(): StorageTransactionState { return this.#state; } diff --git a/packages/crawlee/src/index.ts b/packages/crawlee/src/index.ts index bfd1a4a5fd5e..4fb1d7bc4170 100644 --- a/packages/crawlee/src/index.ts +++ b/packages/crawlee/src/index.ts @@ -1,8 +1,3 @@ -import { log } from '@crawlee/core'; -import { playwrightUtils } from '@crawlee/playwright'; -import { puppeteerUtils } from '@crawlee/puppeteer'; -import { downloadListOfUrls, extractMicrodata, parseOpenGraph, sleep, social } from '@crawlee/utils'; - export * from '@crawlee/core'; export * from '@crawlee/utils'; export * from '@crawlee/basic'; @@ -11,16 +6,4 @@ export * from '@crawlee/http'; export * from '@crawlee/cheerio'; export * from '@crawlee/puppeteer'; export * from '@crawlee/playwright'; -export * from '@crawlee/browser-pool'; export * from '@crawlee/fs-storage'; - -export const utils = { - puppeteer: puppeteerUtils, - playwright: playwrightUtils, - log, - social, - sleep, - downloadListOfUrls, - parseOpenGraph, - extractMicrodata, -}; diff --git a/packages/fs-storage/src/file-system-storage.ts b/packages/fs-storage/src/file-system-storage.ts index 8fffc2269d2a..379a3067f979 100644 --- a/packages/fs-storage/src/file-system-storage.ts +++ b/packages/fs-storage/src/file-system-storage.ts @@ -53,6 +53,8 @@ export interface FileSystemStorageOptions { /** * Optional logger for FileSystemStorageBackend warnings. + * + * @internal */ logger?: CrawleeLogger; @@ -96,12 +98,12 @@ export interface FileSystemStorageOptions { * `teardown` can operate over them), and exposing them through the `@crawlee/types` interfaces. */ export class FileSystemStorageBackend implements storage.StorageBackend { - readonly localDataDirectory: string; - readonly datasetsDirectory: string; - readonly keyValueStoresDirectory: string; - readonly requestQueuesDirectory: string; - readonly logger?: CrawleeLogger; - readonly requestQueueAccess: 'single' | 'shared'; + readonly #localDataDirectory: string; + readonly #datasetsDirectory: string; + readonly #keyValueStoresDirectory: string; + readonly #requestQueuesDirectory: string; + readonly #logger?: CrawleeLogger; + readonly #requestQueueAccess: 'single' | 'shared'; /** `INPUT` plus the configured `inputKey`, deduplicated. */ readonly #inputKeys: string[]; @@ -115,14 +117,14 @@ export class FileSystemStorageBackend implements storage.StorageBackend { fileSystemStorageOptionsSchema, ); - this.logger = logger; - this.requestQueueAccess = requestQueueAccess; + this.#logger = logger; + this.#requestQueueAccess = requestQueueAccess; this.#inputKeys = [...new Set([DEFAULT_INPUT_KEY, inputKey])]; - this.localDataDirectory = localDataDirectory; - this.datasetsDirectory = resolve(this.localDataDirectory, 'datasets'); - this.keyValueStoresDirectory = resolve(this.localDataDirectory, 'key_value_stores'); - this.requestQueuesDirectory = resolve(this.localDataDirectory, 'request_queues'); + this.#localDataDirectory = localDataDirectory; + this.#datasetsDirectory = resolve(this.#localDataDirectory, 'datasets'); + this.#keyValueStoresDirectory = resolve(this.#localDataDirectory, 'key_value_stores'); + this.#requestQueuesDirectory = resolve(this.#localDataDirectory, 'request_queues'); } /** @@ -131,7 +133,7 @@ export class FileSystemStorageBackend implements storage.StorageBackend { * partitions, by including the storage directory in the cache key. */ getStorageBackendCacheKey(): string { - return `FileSystemStorageBackend:${resolve(this.localDataDirectory)}`; + return `FileSystemStorageBackend:${resolve(this.#localDataDirectory)}`; } static #resolveStorageKey(options: { id?: string; name?: string; alias?: string }): { @@ -168,12 +170,12 @@ export class FileSystemStorageBackend implements storage.StorageBackend { const nativeBackend = await ( await importNativeModule() - ).FileSystemDatasetClient.open(id, name, alias, this.localDataDirectory); + ).FileSystemDatasetClient.open(id, name, alias, this.#localDataDirectory); const newStore = await DatasetBackend.create({ name: alias ? undefined : (name ?? cacheKey), cacheKey, nativeBackend, - logger: this.logger, + logger: this.#logger, }); this.#datasetBackendCache.push(newStore); @@ -199,7 +201,7 @@ export class FileSystemStorageBackend implements storage.StorageBackend { id, name, alias, - this.localDataDirectory, + this.#localDataDirectory, // useTestClock — always real wall-clock outside of native tests. undefined, this.#adoptionCandidates(cacheKey === DEFAULT_STORAGE_DIRECTORY), @@ -208,7 +210,7 @@ export class FileSystemStorageBackend implements storage.StorageBackend { name: alias ? undefined : (name ?? cacheKey), cacheKey, nativeBackend, - logger: this.logger, + logger: this.#logger, inputKeys: this.#inputKeys, }); this.#keyValueStoreBackendCache.push(newStore); @@ -235,16 +237,16 @@ export class FileSystemStorageBackend implements storage.StorageBackend { id, name, alias, - this.localDataDirectory, + this.#localDataDirectory, // useTestClock — always real wall-clock outside of native tests. undefined, - this.requestQueueAccess, + this.#requestQueueAccess, ); const newStore = await RequestQueueBackend.create({ name: alias ? undefined : (name ?? cacheKey), cacheKey, nativeBackend, - logger: this.logger, + logger: this.#logger, }); this.#requestQueueBackendCache.push(newStore); @@ -292,15 +294,15 @@ export class FileSystemStorageBackend implements storage.StorageBackend { switch (type) { case 'Dataset': backends = this.#datasetBackendCache; - baseDir = this.datasetsDirectory; + baseDir = this.#datasetsDirectory; break; case 'KeyValueStore': backends = this.#keyValueStoreBackendCache; - baseDir = this.keyValueStoresDirectory; + baseDir = this.#keyValueStoresDirectory; break; case 'RequestQueue': backends = this.#requestQueueBackendCache; - baseDir = this.requestQueuesDirectory; + baseDir = this.#requestQueuesDirectory; break; default: return false; @@ -381,18 +383,18 @@ export class FileSystemStorageBackend implements storage.StorageBackend { async purge(): Promise { await Promise.all([ this.#purgeRunScopedStorages( - this.keyValueStoresDirectory, + this.#keyValueStoresDirectory, async (alias) => this.createKeyValueStoreBackend({ alias }) as Promise, // Only the default store holds the run input, so it is the only one that keeps `INPUT`. async (store, isDefault) => (isDefault ? store.purgeExceptInput() : store.purge()), ), this.#purgeRunScopedStorages( - this.datasetsDirectory, + this.#datasetsDirectory, async (alias) => this.createDatasetBackend({ alias }) as Promise, async (store) => store.purge(), ), this.#purgeRunScopedStorages( - this.requestQueuesDirectory, + this.#requestQueuesDirectory, async (alias) => this.createRequestQueueBackend({ alias }) as Promise, async (store) => store.purge(), ), diff --git a/packages/fs-storage/test/default-storage-layout.test.ts b/packages/fs-storage/test/default-storage-layout.test.ts index 46dd50f4c423..ac6ae805233d 100644 --- a/packages/fs-storage/test/default-storage-layout.test.ts +++ b/packages/fs-storage/test/default-storage-layout.test.ts @@ -3,10 +3,13 @@ import { resolve } from 'node:path'; import { FileSystemStorageBackend } from '@crawlee/fs-storage'; +import { storageLayout } from './storage-layout.js'; + // The default storage lives in `default`. The alias @crawlee/core opens it under is an internal // sentinel, and letting that reach the disk orphans every `storage/` directory an earlier run wrote. describe('the default storage on disk', () => { const tmpLocation = resolve(import.meta.dirname, './tmp/default-storage-layout'); + const { datasetsDirectory, keyValueStoresDirectory, requestQueuesDirectory } = storageLayout(tmpLocation); afterEach(async () => { await rm(tmpLocation, { force: true, recursive: true }); @@ -19,20 +22,17 @@ describe('the default storage on disk', () => { await storage.createKeyValueStoreBackend(); await storage.createRequestQueueBackend(); - expect(await readdir(storage.datasetsDirectory)).toEqual(['default']); - expect(await readdir(storage.keyValueStoresDirectory)).toEqual(['default']); - expect(await readdir(storage.requestQueuesDirectory)).toEqual(['default']); + expect(await readdir(datasetsDirectory)).toEqual(['default']); + expect(await readdir(keyValueStoresDirectory)).toEqual(['default']); + expect(await readdir(requestQueuesDirectory)).toEqual(['default']); }); // The documented way to supply input to a local run: drop a file into the default key-value store // directory by hand. It only works if that directory is the one the default store actually opens. test('reads an INPUT.json placed in the default key-value store directory by hand', async () => { const storage = new FileSystemStorageBackend({ localDataDirectory: tmpLocation }); - await mkdir(resolve(storage.keyValueStoresDirectory, 'default'), { recursive: true }); - await writeFile( - resolve(storage.keyValueStoresDirectory, 'default', 'INPUT.json'), - JSON.stringify({ hello: 'world' }), - ); + await mkdir(resolve(keyValueStoresDirectory, 'default'), { recursive: true }); + await writeFile(resolve(keyValueStoresDirectory, 'default', 'INPUT.json'), JSON.stringify({ hello: 'world' })); const defaultStore = await storage.createKeyValueStoreBackend(); @@ -41,11 +41,8 @@ describe('the default storage on disk', () => { test('keeps a hand-placed INPUT.json across a purge', async () => { const storage = new FileSystemStorageBackend({ localDataDirectory: tmpLocation }); - await mkdir(resolve(storage.keyValueStoresDirectory, 'default'), { recursive: true }); - await writeFile( - resolve(storage.keyValueStoresDirectory, 'default', 'INPUT.json'), - JSON.stringify({ hello: 'world' }), - ); + await mkdir(resolve(keyValueStoresDirectory, 'default'), { recursive: true }); + await writeFile(resolve(keyValueStoresDirectory, 'default', 'INPUT.json'), JSON.stringify({ hello: 'world' })); await storage.purge(); diff --git a/packages/fs-storage/test/key-value-store/adoption.test.ts b/packages/fs-storage/test/key-value-store/adoption.test.ts index a7996774fb33..1ffaa5af7ef6 100644 --- a/packages/fs-storage/test/key-value-store/adoption.test.ts +++ b/packages/fs-storage/test/key-value-store/adoption.test.ts @@ -17,6 +17,9 @@ import type { KeyValueStoreRecord } from '@crawlee/types'; const payload = JSON.stringify({ hello: 'from disk' }); +/** The backend lays key-value stores out under `/key_value_stores`. */ +const storesDirectory = (directory: string) => resolve(directory, 'key_value_stores'); + /** A fresh backend over `directory`, seeded with sidecar-less files in one key-value store. */ async function seedStore( directory: string, @@ -25,7 +28,7 @@ async function seedStore( options: { inputKey?: string } = {}, ): Promise { const storage = new FileSystemStorageBackend({ localDataDirectory: directory, ...options }); - const storeDirectory = resolve(storage.keyValueStoresDirectory, store); + const storeDirectory = resolve(storesDirectory(directory), store); await mkdir(storeDirectory, { recursive: true }); for (const [file, content] of Object.entries(files)) { await writeFile(resolve(storeDirectory, file), content); @@ -65,7 +68,7 @@ describe('a sidecar-less run-input file in the default store', () => { await store.deleteValue('INPUT'); expect(await store.getValue('INPUT')).toBeUndefined(); - expect(await readdir(resolve(storage.keyValueStoresDirectory, 'default'))).not.toContain('INPUT.json'); + expect(await readdir(resolve(storesDirectory(tmpLocation), 'default'))).not.toContain('INPUT.json'); }); test('reads as bytes when it has no extension', async () => { @@ -105,7 +108,7 @@ describe('a sidecar-less run-input file in the default store', () => { const storage = await seedStore(tmpLocation, 'default', {}); const store = await storage.createKeyValueStoreBackend(); await store.setValue({ key: 'INPUT', value: 'tracked', contentType: 'text/plain; charset=utf-8' }); - await writeFile(resolve(storage.keyValueStoresDirectory, 'default', 'INPUT.json'), payload); + await writeFile(resolve(storesDirectory(tmpLocation), 'default', 'INPUT.json'), payload); // Reopening must not rebind the key: a stray file is not allowed to take over a record the // run wrote itself, and adopting it under its own filename would make `INPUT.json` a second @@ -250,7 +253,7 @@ describe('purging a store with adopted records', () => { const store = await storage.createKeyValueStoreBackend(); expect((await store.listKeys()).items.map((item) => item.key)).toEqual(['INPUT', inputKey]); expect((await store.getValue('INPUT'))?.value.toString()).toBe(payload); - expect(await readdir(resolve(storage.keyValueStoresDirectory, 'default'))).not.toContain('leftover.json'); + expect(await readdir(resolve(storesDirectory(tmpLocation), 'default'))).not.toContain('leftover.json'); }); test('drops the input of a non-default store', async () => { @@ -259,6 +262,6 @@ describe('purging a store with adopted records', () => { await storage.purge(); - expect(await readdir(resolve(storage.keyValueStoresDirectory, 'other'))).not.toContain('INPUT.json'); + expect(await readdir(resolve(storesDirectory(tmpLocation), 'other'))).not.toContain('INPUT.json'); }); }); diff --git a/packages/fs-storage/test/request-queue/request-queue-access.test.ts b/packages/fs-storage/test/request-queue/request-queue-access.test.ts index f3b8bc1c3efa..710188d82f78 100644 --- a/packages/fs-storage/test/request-queue/request-queue-access.test.ts +++ b/packages/fs-storage/test/request-queue/request-queue-access.test.ts @@ -15,16 +15,6 @@ describe('FileSystemStorageBackend requestQueueAccess', () => { await rm(tmpLocation, { force: true, recursive: true }); }); - test("defaults to 'single'", () => { - const storage = new FileSystemStorageBackend({ localDataDirectory: tmpLocation }); - expect(storage.requestQueueAccess).toBe('single'); - }); - - test("respects an explicit 'shared'", () => { - const storage = new FileSystemStorageBackend({ localDataDirectory: tmpLocation, requestQueueAccess: 'shared' }); - expect(storage.requestQueueAccess).toBe('shared'); - }); - // Seed a queue with two requests, fetch (lock) one without handling it or tearing down — leaving a // dangling in-progress lock on disk, exactly the "process died mid-flight" situation. async function seedQueueWithDanglingLock(dir: string) { @@ -40,7 +30,7 @@ describe('FileSystemStorageBackend requestQueueAccess', () => { return locked!; } - test("'single' (default): reopening preserves contents but relinquishes the dangling lock", async () => { + test("'single': reopening preserves contents but relinquishes the dangling lock", async () => { const dir = resolve(tmpLocation, 'single'); const locked = await seedQueueWithDanglingLock(dir); @@ -63,6 +53,21 @@ describe('FileSystemStorageBackend requestQueueAccess', () => { expect(reFetched?.url).toBe(locked.url); }); + // The default is `'single'`, which is only observable through the reclaim behavior: with no + // `requestQueueAccess` given at all, the dangling lock is relinquished exactly as it is above. + test("defaults to 'single'", async () => { + const dir = resolve(tmpLocation, 'default'); + const locked = await seedQueueWithDanglingLock(dir); + + const reopened = new FileSystemStorageBackend({ localDataDirectory: dir }); + const queue = await reopened.createRequestQueueBackend({ name: 'default' }); + + const a = await queue.fetchNextRequest(); + const b = await queue.fetchNextRequest(); + expect([a?.uniqueKey, b?.uniqueKey].sort()).toStrictEqual(['1', '2']); + expect((await queue.getRequest(locked.uniqueKey))?.url).toBe(locked.url); + }); + test("'shared': reopening keeps the dangling lock (concurrency-safe mode)", async () => { const dir = resolve(tmpLocation, 'shared'); await seedQueueWithDanglingLock(dir); diff --git a/packages/fs-storage/test/storage-layout.ts b/packages/fs-storage/test/storage-layout.ts new file mode 100644 index 000000000000..109f4b026f02 --- /dev/null +++ b/packages/fs-storage/test/storage-layout.ts @@ -0,0 +1,15 @@ +import { resolve } from 'node:path'; + +/** + * The on-disk directory layout a `FileSystemStorageBackend` writes under its configured + * `localDataDirectory`. The layout is part of the documented contract, so the tests compute it here + * instead of reading it back off the backend instance — that way they assert the layout rather than + * echo whatever the backend happens to have resolved. + */ +export function storageLayout(localDataDirectory: string) { + return { + datasetsDirectory: resolve(localDataDirectory, 'datasets'), + keyValueStoresDirectory: resolve(localDataDirectory, 'key_value_stores'), + requestQueuesDirectory: resolve(localDataDirectory, 'request_queues'), + }; +} diff --git a/packages/got-scraping-client/README.md b/packages/got-scraping-client/README.md index 68fd8c2fc770..b1cf1de15084 100644 --- a/packages/got-scraping-client/README.md +++ b/packages/got-scraping-client/README.md @@ -10,7 +10,7 @@ Simply pass the `GotScrapingHttpClient` instance to the `httpClient` option of t ```typescript import { CheerioCrawler, Dictionary } from '@crawlee/cheerio'; -import { GotScrapingHttpClient, Browser } from '@crawlee/got-scraping-client'; +import { GotScrapingHttpClient } from '@crawlee/got-scraping-client'; const crawler = new CheerioCrawler({ httpClient: new GotScrapingHttpClient(), diff --git a/packages/got-scraping-client/src/index.ts b/packages/got-scraping-client/src/index.ts index a267f6de5f66..de8689b8e3cd 100644 --- a/packages/got-scraping-client/src/index.ts +++ b/packages/got-scraping-client/src/index.ts @@ -11,13 +11,13 @@ export class GotScrapingHttpClient extends BaseHttpClient { * Type guard that validates the HTTP method (excluding CONNECT). * @param request - The HTTP request to validate */ - private validateRequest( + #validateRequest( request: Request, ): request is Request & { method: Exclude } { return !['CONNECT', 'connect'].includes(request.method!); } - private *iterateHeaders( + *#iterateHeaders( headers: Record, ): Generator<[string, string], void, unknown> { for (const [key, value] of Object.entries(headers)) { @@ -30,14 +30,14 @@ export class GotScrapingHttpClient extends BaseHttpClient { } } - private parseHeaders(headers: Record): Headers { - return new Headers([...this.iterateHeaders(headers)]); + #parseHeaders(headers: Record): Headers { + return new Headers([...this.#iterateHeaders(headers)]); } - override async fetch(request: Request, options?: RequestInit & CustomFetchOptions): Promise { + protected override async fetch(request: Request, options?: RequestInit & CustomFetchOptions): Promise { const { proxyUrl, redirect, ignoreTlsErrors } = options ?? {}; - if (!this.validateRequest(request)) { + if (!this.#validateRequest(request)) { throw new Error(`The HTTP method CONNECT is not supported by the GotScrapingHttpClient.`); } @@ -52,7 +52,7 @@ export class GotScrapingHttpClient extends BaseHttpClient { ...(ignoreTlsErrors ? { https: { rejectUnauthorized: false } } : {}), }); - const responseHeaders = this.parseHeaders(gotResult.headers); + const responseHeaders = this.#parseHeaders(gotResult.headers); return new ResponseWithUrl(new Uint8Array(gotResult.rawBody), { headers: responseHeaders, diff --git a/packages/http-client/src/base-http-client.ts b/packages/http-client/src/base-http-client.ts index 9b13ad047acc..9b496bcaa120 100644 --- a/packages/http-client/src/base-http-client.ts +++ b/packages/http-client/src/base-http-client.ts @@ -62,7 +62,7 @@ export abstract class BaseHttpClient implements BaseHttpClientInterface { */ protected abstract fetch(input: Request, init?: RequestInit & CustomFetchOptions): Promise; - private async applyCookies(request: Request, cookieJar: CookieJar): Promise { + async #applyCookies(request: Request, cookieJar: CookieJar): Promise { try { const requestCookies = request.headers.get('cookie') ?? ''; @@ -97,7 +97,7 @@ export abstract class BaseHttpClient implements BaseHttpClientInterface { return request; } - private async setCookies(response: Response, cookieJar: CookieJar): Promise { + async #setCookies(response: Response, cookieJar: CookieJar): Promise { const setCookieHeaders = response.headers.getSetCookie(); for (const header of setCookieHeaders) { @@ -109,7 +109,7 @@ export abstract class BaseHttpClient implements BaseHttpClientInterface { } } - private async resolveRequestContext(options?: SendRequestOptions): Promise<{ + async #resolveRequestContext(options?: SendRequestOptions): Promise<{ proxyUrl?: string; cookieJar: CookieJar; signal?: AbortSignal; @@ -118,7 +118,7 @@ export abstract class BaseHttpClient implements BaseHttpClientInterface { }> { const proxyUrl = options?.proxyUrl ?? options?.session?.proxyInfo?.url; const cookieJar = options?.cookieJar ?? options?.session?.cookieJar ?? (await this.#createDefaultCookieJar()); - const signal = this.createAbortSignal(options?.signal, options?.timeoutMillis); + const signal = this.#createAbortSignal(options?.signal, options?.timeoutMillis); return { proxyUrl, cookieJar, @@ -133,7 +133,7 @@ export abstract class BaseHttpClient implements BaseHttpClientInterface { return new ToughCookieJar(); } - private createAbortSignal(signal?: AbortSignal, timeoutMillis?: number): AbortSignal | undefined { + #createAbortSignal(signal?: AbortSignal, timeoutMillis?: number): AbortSignal | undefined { if (signal && timeoutMillis) { return AbortSignal.any([signal, AbortSignal.timeout(timeoutMillis)]); } @@ -143,12 +143,12 @@ export abstract class BaseHttpClient implements BaseHttpClientInterface { return timeoutMillis ? AbortSignal.timeout(timeoutMillis) : undefined; } - private isRedirect(response: Response): boolean { + #isRedirect(response: Response): boolean { const status = response.status; return status >= 300 && status < 400 && !!response.headers.get('location'); } - private buildRedirectRequest(currentRequest: Request, response: Response, initialRequest: Request): Request { + #buildRedirectRequest(currentRequest: Request, response: Response, initialRequest: Request): Request { const location = response.headers.get('location')!; const nextUrl = new URL(location, response.url ?? currentRequest.url); @@ -187,11 +187,12 @@ export abstract class BaseHttpClient implements BaseHttpClientInterface { let currentRequest = initialRequest; let redirectCount = 0; - const { proxyUrl, cookieJar, signal, fingerprint, ignoreTlsErrors } = await this.resolveRequestContext(options); + const { proxyUrl, cookieJar, signal, fingerprint, ignoreTlsErrors } = + await this.#resolveRequestContext(options); currentRequest = initialRequest.clone(); while (true) { - await this.applyCookies(currentRequest, cookieJar); + await this.#applyCookies(currentRequest, cookieJar); const response = await this.fetch(currentRequest, { signal, @@ -202,13 +203,13 @@ export abstract class BaseHttpClient implements BaseHttpClientInterface { redirect: 'manual', }); - await this.setCookies(response, cookieJar); + await this.#setCookies(response, cookieJar); - if (this.isRedirect(response)) { + if (this.#isRedirect(response)) { if (redirectCount++ >= maxRedirects) { throw new Error(`Too many redirects (${maxRedirects}) while requesting ${currentRequest.url}`); } - currentRequest = this.buildRedirectRequest(currentRequest, response, initialRequest); + currentRequest = this.#buildRedirectRequest(currentRequest, response, initialRequest); continue; } diff --git a/packages/http-client/src/fetch-http-client.ts b/packages/http-client/src/fetch-http-client.ts index f4be3c298e2e..f5fdd82e5fd9 100644 --- a/packages/http-client/src/fetch-http-client.ts +++ b/packages/http-client/src/fetch-http-client.ts @@ -15,7 +15,7 @@ export class FetchHttpClient extends BaseHttpClient { this.#logger = options?.logger; } - override async fetch(request: Request, options?: RequestInit & CustomFetchOptions): Promise { + protected override async fetch(request: Request, options?: RequestInit & CustomFetchOptions): Promise { if (options?.ignoreTlsErrors) { this.#logger?.warningOnce( 'FetchHttpClient cannot disable TLS certificate verification, the `ignoreTlsErrors` option is ignored. ' + diff --git a/packages/http-client/src/index.ts b/packages/http-client/src/index.ts index 104f8fbf5057..c4f662ceec55 100644 --- a/packages/http-client/src/index.ts +++ b/packages/http-client/src/index.ts @@ -1,3 +1,3 @@ export { BaseHttpClient, type CustomFetchOptions } from './base-http-client.js'; -export { ResponseWithUrl, type IResponseWithUrl } from './response.js'; +export { ResponseWithUrl } from './response.js'; export { FetchHttpClient } from './fetch-http-client.js'; diff --git a/packages/http-client/src/response.ts b/packages/http-client/src/response.ts index 15268b0f3392..d223f0a951b7 100644 --- a/packages/http-client/src/response.ts +++ b/packages/http-client/src/response.ts @@ -1,7 +1,3 @@ -export interface IResponseWithUrl extends Response { - url: string; -} - // See https://github.com/nodejs/undici/blob/d7707ee8fd5da2d0cc64b5fae421b965faf803c8/lib/web/fetch/constants.js#L6 const nullBodyStatus = [101, 204, 205, 304]; @@ -10,8 +6,8 @@ const nullBodyStatus = [101, 204, 205, 304]; * * This class extends `Response` from `fetch` API and is fully compatible with this. */ -export class ResponseWithUrl extends Response implements IResponseWithUrl { - override url: string; +export class ResponseWithUrl extends Response { + override readonly url: string; constructor(body: BodyInit | null, init: ResponseInit & { url?: string }) { const bodyParsed = nullBodyStatus.includes(init.status ?? 200) ? null : body; diff --git a/packages/http-crawler/src/internals/file-download.ts b/packages/http-crawler/src/internals/file-download.ts index 02cbd25a915e..389ca590ec16 100644 --- a/packages/http-crawler/src/internals/file-download.ts +++ b/packages/http-crawler/src/internals/file-download.ts @@ -9,7 +9,6 @@ import type { Dictionary } from '@crawlee/types'; import type { ErrorHandler, GetUserDataFromRequest, - InternalHttpHook, RequestHandler, RouterHandler, RouterRoutes, @@ -26,10 +25,6 @@ export type FileDownloadErrorHandler< ContextExtension = Dictionary, > = ErrorHandler & ContextExtension>; -export type FileDownloadHook< - UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler -> = InternalHttpHook>; - export interface FileDownloadCrawlingContext< UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler > extends CrawlingContext { @@ -42,88 +37,6 @@ export type FileDownloadRequestHandler< UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler > = RequestHandler>; -/** - * Creates a transform stream that throws an error if the source data speed is below the specified minimum speed. - * This `Transform` checks the amount of data every `checkProgressInterval` milliseconds. - * If the stream has received less than `minSpeedKbps * historyLengthMs / 1000` bytes in the last `historyLengthMs` milliseconds, - * it will throw an error. - * - * Can be used e.g. to abort a download if the network speed is too slow. - * @returns Transform stream that monitors the speed of the incoming data. - */ -export function MinimumSpeedStream({ - minSpeedKbps, - historyLengthMs = 10e3, - checkProgressInterval: checkProgressIntervalMs = 5e3, -}: { - minSpeedKbps: number; - historyLengthMs?: number; - checkProgressInterval?: number; -}): Transform { - let snapshots: { timestamp: number; bytes: number }[] = []; - - const checkInterval = setInterval(() => { - const now = Date.now(); - - snapshots = snapshots.filter((snapshot) => now - snapshot.timestamp < historyLengthMs); - const totalBytes = snapshots.reduce((acc, snapshot) => acc + snapshot.bytes, 0); - const elapsed = (now - (snapshots[0]?.timestamp ?? 0)) / 1000; - - if (totalBytes / 1024 / elapsed < minSpeedKbps) { - clearInterval(checkInterval); - stream.emit('error', new Error(`Stream speed too slow, aborting...`)); - } - }, checkProgressIntervalMs); - - const stream = new Transform({ - transform: (chunk, _, callback) => { - snapshots.push({ timestamp: Date.now(), bytes: chunk.length }); - callback(null, chunk); - }, - final: (callback) => { - clearInterval(checkInterval); - callback(); - }, - }); - - return stream; -} - -/** - * Creates a transform stream that logs the progress of the incoming data. - * This `Transform` calls the `logProgress` function every `loggingInterval` milliseconds with the number of bytes received so far. - * - * Can be used e.g. to log the progress of a download. - * @returns Transform stream logging the progress of the incoming data. - */ -export function ByteCounterStream({ - logTransferredBytes, - loggingInterval = 5000, -}: { - logTransferredBytes: (transferredBytes: number) => void; - loggingInterval?: number; -}): Transform { - let transferredBytes = 0; - let lastLogTime = Date.now(); - - return new Transform({ - transform: (chunk, _, callback) => { - transferredBytes += chunk.length; - - if (Date.now() - lastLogTime > loggingInterval) { - lastLogTime = Date.now(); - logTransferredBytes(transferredBytes); - } - - callback(null, chunk); - }, - flush: (callback) => { - logTransferredBytes(transferredBytes); - callback(); - }, - }); -} - /** * Provides a framework for downloading files in parallel using plain HTTP requests. The URLs to download are fed either from a static list of URLs or they can be added on the fly from another crawler. * @@ -131,11 +44,11 @@ export function ByteCounterStream({ * However, it doesn't parse the content - if you need to e.g. extract data from the downloaded files, * you might need to use {@apilink CheerioCrawler}, {@apilink PuppeteerCrawler} or {@apilink PlaywrightCrawler} instead. * - * `FileCrawler` downloads each URL using a plain HTTP request and then invokes the user-provided {@apilink FileDownloadOptions.requestHandler} where the user can specify what to do with the downloaded data. + * `FileCrawler` downloads each URL using a plain HTTP request and then invokes the user-provided {@apilink BasicCrawlerOptions.requestHandler} where the user can specify what to do with the downloaded data. * - * The source URLs are represented using {@apilink Request} objects that are fed from the {@apilink IRequestManager|request manager} provided via the {@apilink FileDownloadOptions.requestManager|`requestManager`} constructor option (a {@apilink RequestQueue} is itself a request manager). To read from a read-only source such as a {@apilink RequestList} while still being able to enqueue new requests, combine it with a queue into a {@apilink RequestManagerTandem} via {@apilink IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the result as `requestManager`. + * The source URLs are represented using {@apilink Request} objects that are fed from the {@apilink IRequestManager|request manager} provided via the {@apilink BasicCrawlerOptions.requestManager|`requestManager`} constructor option (a {@apilink RequestQueue} is itself a request manager). To read from a read-only source such as a {@apilink RequestList} while still being able to enqueue new requests, combine it with a queue into a {@apilink RequestManagerTandem} via {@apilink IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the result as `requestManager`. * - * > The {@apilink FileDownloadOptions.requestList|`requestList`} and {@apilink FileDownloadOptions.requestQueue|`requestQueue`} options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat. + * > The {@apilink BasicCrawlerOptions.requestList|`requestList`} and {@apilink BasicCrawlerOptions.requestQueue|`requestQueue`} options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat. * * The crawler finishes when there are no more {@apilink Request} objects to crawl. * @@ -172,11 +85,11 @@ export class FileDownload extends BasicCrawler { constructor(options: BasicCrawlerOptions = {}) { super({ ...options, - contextPipelineBuilder: () => this.buildContextPipeline(), + contextPipelineBuilder: () => this.#buildContextPipeline(), }); } - protected override buildContextPipeline(): ContextPipeline { + #buildContextPipeline(): ContextPipeline { return super.buildContextPipeline().compose({ action: async (context) => this.initiateDownload(context), cleanup: async (context) => { diff --git a/packages/http-crawler/src/internals/http-crawler.ts b/packages/http-crawler/src/internals/http-crawler.ts index 275e63f9738b..699c8a2c96c5 100644 --- a/packages/http-crawler/src/internals/http-crawler.ts +++ b/packages/http-crawler/src/internals/http-crawler.ts @@ -50,7 +50,8 @@ const HTML_AND_XML_MIME_TYPES = ['text/html', 'text/xml', 'application/xhtml+xml const APPLICATION_JSON_MIME_TYPE = 'application/json'; /** * A higher starting concurrency and a relaxed event loop signal, since HTTP-only crawling barely touches the event - * loop. {@apilink HttpCrawler} folds these into the {@apilink ConcurrencySystem} it builds by default. + * loop. {@apilink HttpCrawler} folds these into the {@apilink ConcurrencySystem} it builds by default, with your own + * concurrency shortcuts (`minConcurrency`, `maxConcurrency`, `maxRequestsPerMinute`) kept on top. * * A {@apilink BasicCrawlerOptions.concurrencySystem|`concurrencySystem`} you supply yourself replaces that default * wholesale, tuning included, so spread these options in if you want to keep it: @@ -142,9 +143,7 @@ export interface HttpCrawlerOptions< * ] * ``` */ - postNavigationHooks?: (( - crawlingContext: CrawlingContextWithResponse & ContextExtension, - ) => Awaitable>)[]; + postNavigationHooks?: InternalHttpHook[]; /** * An array of [MIME types](https://developer.mozilla.org/en-US/docs/Web/HTTP/Basics_of_HTTP/MIME_types/Complete_list_of_MIME_types) @@ -190,11 +189,6 @@ export type InternalHttpHook = ( crawlingContext: Context & ContextExtension, ) => Awaitable>; -export type HttpHook< - UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler - JSONData extends JsonValue = any, // with default to Dictionary we cant use a typed router in untyped crawler -> = InternalHttpHook>; - interface CrawlingContextWithResponse< UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler > extends CrawlingContext { @@ -209,10 +203,6 @@ interface CrawlingContextWithResponse< response: Response; } -type InternalHttpPostNavigationHook = ( - crawlingContext: CrawlingContextWithResponse, -) => Awaitable>; - export interface InternalHttpCrawlingContext< UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler JSONData extends JsonValue = any, // with default to Dictionary we cant use a typed router in untyped crawler @@ -236,7 +226,11 @@ export interface InternalHttpCrawlingContext< contentType: { type: string; encoding: BufferEncoding }; /** - * Wait for an element matching the selector to appear. Timeout is ignored. + * Wait for an element matching the selector to appear. + * + * `HttpCrawler` and {@apilink CheerioCrawler} parse a response that is already fully downloaded, so there is + * nothing to wait for and `timeoutMs` is ignored. {@apilink JSDOMCrawler} and {@apilink LinkeDOMCrawler} poll for + * the selector instead, with `timeoutMs` defaulting to 5s. * * **Example usage:** * ```ts @@ -251,7 +245,12 @@ export interface InternalHttpCrawlingContext< /** * Returns Cheerio handle for `page.content()`, allowing to work with the data same way as with {@apilink CheerioCrawler}. - * When provided with the `selector` argument, it will throw if it's not available. + * This is here to unify the crawler API, so they all have this handy method - in {@apilink CheerioCrawler} it has + * the same return type as the `$` context property, so use it only if you are abstracting your workflow to + * support different context types in one handler. + * + * When provided with the `selector` argument, it will throw if it's not available. {@apilink JSDOMCrawler} and + * {@apilink LinkeDOMCrawler} wait for the selector first, with `timeoutMs` defaulting to 5s. * * **Example usage:** * ```ts @@ -355,7 +354,7 @@ export class HttpCrawler< // concrete crawling context, which does not statically carry `ContextExtension`. The members // added by `extendContext` are present at runtime regardless. #preNavigationHooks: InternalHttpHook[]; - #postNavigationHooks: InternalHttpPostNavigationHook[]; + #postNavigationHooks: InternalHttpHook[]; #saveResponseCookies: boolean; #navigationTimeoutMillis: number; #ignoreTlsErrors: boolean; @@ -431,22 +430,18 @@ export class HttpCrawler< this.#preNavigationHooks = preNavigationHooks as InternalHttpHook[]; this.#postNavigationHooks = [ ({ request, response }) => this.abortDownloadOfBody(request, response!), - ...(postNavigationHooks as InternalHttpPostNavigationHook[]), + ...(postNavigationHooks as InternalHttpHook[]), ]; this.#saveResponseCookies = saveResponseCookies; } + /** @internal */ protected override getNavigationTimeoutMillis(): number { return this.#navigationTimeoutMillis; } - /** - * Folds {@apilink HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS} into the default system, keeping the user's - * concurrency shortcuts on top. Not called for a supplied - * {@apilink BasicCrawlerOptions.concurrencySystem|`concurrencySystem`} — spread the constant into it yourself to - * keep the tuning. - */ + /** @internal */ protected override createDefaultConcurrencySystem(options: ConcurrencySystemOptions): ConcurrencySystem { return super.createDefaultConcurrencySystem({ ...HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS, @@ -454,6 +449,7 @@ export class HttpCrawler< }); } + /** @internal */ protected override buildContextPipeline(): ContextPipeline { // When navigation is skipped, `prepareHttpRequest` has already installed throwing getters for // the response-derived members, so the guarded action is bypassed and the context left untouched. @@ -669,13 +665,13 @@ export class HttpCrawler< private async handleBlockedRequestByContent(crawlingContext: InternalHttpCrawlingContext): Promise<{}> { if (this.retryOnBlocked) { - const error = await this.isRequestBlocked(crawlingContext); + const error = await this.#isRequestBlocked(crawlingContext); if (error) throw new SessionError(error); } return {}; } - protected async isRequestBlocked(crawlingContext: InternalHttpCrawlingContext): Promise { + async #isRequestBlocked(crawlingContext: InternalHttpCrawlingContext): Promise { if (HTML_AND_XML_MIME_TYPES.includes(crawlingContext.contentType.type)) { const $ = await crawlingContext.parseWithCheerio(); @@ -699,7 +695,7 @@ export class HttpCrawler< * received content type matches text/html, application/xml, application/xhtml+xml. */ private async requestFunction({ request, session, proxyUrl }: RequestFunctionOptions): Promise { - const opts = this.getRequestOptions(request, session, proxyUrl); + const opts = this.getRequestOptions(request, proxyUrl); try { return await this.requestAsBrowser(opts, session); @@ -710,7 +706,7 @@ export class HttpCrawler< } if (this.isProxyError(e as Error)) { - throw new SessionError(this.getMessageFromError(e as Error) as string); + throw new SessionError(this.getMessageFromError(e as Error)); } else { throw e; } @@ -772,13 +768,12 @@ export class HttpCrawler< /** * Combines the provided `requestOptions` with mandatory (non-overridable) values. */ - private getRequestOptions(request: CrawleeRequest, session: ISession, proxyUrl?: string) { + private getRequestOptions(request: CrawleeRequest, proxyUrl?: string) { const requestOptions = { url: request.url, method: request.method, proxyUrl, timeout: this.#navigationTimeoutMillis, - sessionToken: session, headers: request.headers, body: undefined as string | undefined, }; diff --git a/packages/impit-client/src/index.ts b/packages/impit-client/src/index.ts index 89549869e837..596ecff857d0 100644 --- a/packages/impit-client/src/index.ts +++ b/packages/impit-client/src/index.ts @@ -54,7 +54,7 @@ export class ImpitHttpClient extends BaseHttpClient { */ #impitBrowserByFingerprint = new WeakMap(); - private getClient(options: ImpitOptions): Impit { + #getClient(options: ImpitOptions): Impit { if (!this.#cacheClients) { return new Impit(options); } @@ -90,12 +90,12 @@ export class ImpitHttpClient extends BaseHttpClient { /** * @inheritDoc */ - async fetch(request: Request, options?: RequestInit & CustomFetchOptions): Promise { + protected override async fetch(request: Request, options?: RequestInit & CustomFetchOptions): Promise { const { proxyUrl, redirect, signal, fingerprint, ignoreTlsErrors } = options ?? {}; - const impitBrowser = this.resolveImpitBrowser(fingerprint); + const impitBrowser = this.#resolveImpitBrowser(fingerprint); - const impit = this.getClient({ + const impit = this.#getClient({ ...this.#impitOptions, ...(impitBrowser ? { browser: impitBrowser } : {}), // The per-request flag (from the crawler option or a MITM proxy session) @@ -111,7 +111,7 @@ export class ImpitHttpClient extends BaseHttpClient { return new ResponseWithUrl(response.body, response); } - private resolveImpitBrowser(fingerprint?: SessionFingerprint): ImpitBrowser | undefined { + #resolveImpitBrowser(fingerprint?: SessionFingerprint): ImpitBrowser | undefined { if (!fingerprint?.browser) return undefined; const cached = this.#impitBrowserByFingerprint.get(fingerprint); diff --git a/packages/playwright-crawler/src/index.ts b/packages/playwright-crawler/src/index.ts index d2643a3bf042..a89408980a7b 100644 --- a/packages/playwright-crawler/src/index.ts +++ b/packages/playwright-crawler/src/index.ts @@ -1,7 +1,8 @@ export * from '@crawlee/browser'; export * from './internals/playwright-browser-pool.js'; export * from './internals/playwright-crawler.js'; -export * from './internals/playwright-launcher.js'; +export { launchPlaywright } from './internals/playwright-launcher.js'; +export type { PlaywrightLaunchContext } from './internals/playwright-launcher.js'; export * from './internals/adaptive-playwright-crawler.js'; export { RenderingTypePredictor } from './internals/utils/rendering-type-prediction.js'; diff --git a/packages/playwright-crawler/src/internals/enqueue-links/click-elements.ts b/packages/playwright-crawler/src/internals/enqueue-links/click-elements.ts index e5cbcd9694d4..e2ffca888193 100644 --- a/packages/playwright-crawler/src/internals/enqueue-links/click-elements.ts +++ b/packages/playwright-crawler/src/internals/enqueue-links/click-elements.ts @@ -6,7 +6,6 @@ import type { RequestTransform, SkippedRequestCallback, UrlPatternInput, - UrlPatternObject, } from '@crawlee/browser'; import { applyRequestTransform, @@ -239,8 +238,8 @@ export async function enqueueLinksByClickingElements( const maxWaitForPageIdleMillis = maxWaitForPageIdleSecs * 1000; const hasOnSkippedRequest = onSkippedRequest !== undefined; - const urlExcludePatternObjects: UrlPatternObject[] = exclude?.length ? constructUrlPatternObjects(exclude) : []; - const urlPatternObjects: UrlPatternObject[] = include?.length ? constructUrlPatternObjects(include) : []; + const urlExcludePatternObjects = exclude?.length ? constructUrlPatternObjects(exclude) : []; + const urlPatternObjects = include?.length ? constructUrlPatternObjects(include) : []; const interceptedRequests = await clickElementsAndInterceptNavigationRequests({ page, diff --git a/packages/playwright-crawler/src/internals/playwright-crawler.ts b/packages/playwright-crawler/src/internals/playwright-crawler.ts index 1f35c39d4ffb..f29d18b13125 100644 --- a/packages/playwright-crawler/src/internals/playwright-crawler.ts +++ b/packages/playwright-crawler/src/internals/playwright-crawler.ts @@ -5,16 +5,15 @@ import type { ContextPipeline, CrawlingContext, GetUserDataFromRequest, - RequestHandler, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas, } from '@crawlee/browser'; -import { assertBrowserPoolNotConfigured, BrowserCrawler, RequestState, Router, serviceLocator } from '@crawlee/browser'; +import { BrowserCrawler, RequestState, Router, serviceLocator } from '@crawlee/browser'; import type { Dictionary } from '@crawlee/types'; -import { parseArgument, schemas } from '@crawlee/utils/internal'; -import type { Download, LaunchOptions, Page, Response } from 'playwright'; +import { assertBrowserPoolNotConfigured, parseArgument, schemas } from '@crawlee/utils/internal'; +import type { Download, Page, Response } from 'playwright'; import { z } from 'zod'; import type { EnqueueLinksByClickingElementsOptions } from './enqueue-links/click-elements.js'; @@ -29,7 +28,18 @@ import type { PlaywrightContextUtils, SaveSnapshotOptions, } from './utils/playwright-utils.js'; -import { gotoExtended, playwrightUtils } from './utils/playwright-utils.js'; +import { + blockRequests, + compileScript, + enqueueLinksByClickingElements, + gotoExtended, + handleCloudflareChallenge, + infiniteScroll, + injectFile, + injectJQuery, + parseWithCheerio, + saveSnapshot, +} from './utils/playwright-utils.js'; export type PlaywrightGotoOptions = NonNullable[1]>; @@ -69,30 +79,6 @@ export interface PlaywrightCrawlerOptions< */ headless?: boolean; - /** - * Function that is called to process each request. - * - * The function receives the {@apilink PlaywrightCrawlingContext} as an argument, where: - * - `request` is an instance of the {@apilink Request} object with details about the URL to open, HTTP method etc. - * - `page` is an instance of the `Playwright` - * [`Page`](https://playwright.dev/docs/api/class-page) - * - `response` is an instance of the `Playwright` - * [`Response`](https://playwright.dev/docs/api/class-response), - * which is the main resource response as returned by `page.goto(request.url)`. - * - * The function must return a promise, which is then awaited by the crawler. - * - * If the function throws an exception, the crawler will try to re-crawl the - * request later, up to `option.maxRequestRetries` times. - * If all the retries fail, the crawler calls the function - * provided to the `failedRequestHandler` parameter. - * To make this work, you should **always** - * let your function throw exceptions rather than catch them. - * The exceptions are logged to the request using the - * {@apilink Request.pushErrorMessage} function. - */ - requestHandler?: RouterHandler | RequestHandler; - /** * Async functions that are sequentially evaluated before the navigation. Good for setting additional cookies * or browser properties before navigation. The function receives the `crawlingContext`; the options object @@ -212,7 +198,6 @@ export class PlaywrightCrawler< > extends BrowserCrawler< Page, Response, - LaunchOptions, PlaywrightCrawlingContext, ContextExtension, ExtendedContext, @@ -224,8 +209,8 @@ export class PlaywrightCrawler< */ protected static override optionsShape = { ...BrowserCrawler.optionsShape, + launchContext: schemas.anyObject.default(() => ({})), headless: z.boolean().optional(), - launcher: schemas.anyObject.optional(), }; /** @internal */ @@ -264,7 +249,6 @@ export class PlaywrightCrawler< Routes, StatisticStateExtension >), - launchContext, configuration, browserPoolBuilder: (remoteBrowser) => remoteBrowser @@ -296,46 +280,44 @@ export class PlaywrightCrawler< return { injectFile: async (filePath: string, options?: InjectFileOptions) => - playwrightUtils.injectFile(context.page, filePath, options), + injectFile(context.page, filePath, options), injectJQuery: async () => { if (context.request.state === RequestState.BEFORE_NAV) { context.log.warning( 'Using injectJQuery() in preNavigationHooks leads to unstable results. Use it in a postNavigationHook or a requestHandler instead.', ); - await playwrightUtils.injectJQuery(context.page); + await injectJQuery(context.page); return; } - await playwrightUtils.injectJQuery(context.page, { surviveNavigations: false }); + await injectJQuery(context.page, { surviveNavigations: false }); }, - blockRequests: async (options?: BlockRequestsOptions) => - playwrightUtils.blockRequests(context.page, options), + blockRequests: async (options?: BlockRequestsOptions) => blockRequests(context.page, options), waitForSelector, parseWithCheerio: async (selector?: string, timeoutMs = 5_000) => { if (selector) { await waitForSelector(selector, timeoutMs); } - return playwrightUtils.parseWithCheerio(context.page, this.ignoreShadowRoots, this.ignoreIframes); + return parseWithCheerio(context.page, this.ignoreShadowRoots, this.ignoreIframes); }, - infiniteScroll: async (options?: InfiniteScrollOptions) => - playwrightUtils.infiniteScroll(context.page, options), + infiniteScroll: async (options?: InfiniteScrollOptions) => infiniteScroll(context.page, options), listDownloads: async () => downloads, saveSnapshot: async (options?: SaveSnapshotOptions) => - playwrightUtils.saveSnapshot(context.page, { + saveSnapshot(context.page, { ...options, configuration: serviceLocator.getConfiguration(), }), enqueueLinksByClickingElements: async ( options: Omit, ) => - playwrightUtils.enqueueLinksByClickingElements({ + enqueueLinksByClickingElements({ ...options, page: context.page, requestManager: this.requestManager!, }), - compileScript: (scriptString: string, ctx?: Dictionary) => playwrightUtils.compileScript(scriptString, ctx), + compileScript: (scriptString: string, ctx?: Dictionary) => compileScript(scriptString, ctx), handleCloudflareChallenge: async (options?: HandleCloudflareChallengeOptions) => { - return playwrightUtils.handleCloudflareChallenge(context.page, context.request.url, options); + return handleCloudflareChallenge(context.page, context.request.url, options); }, }; } diff --git a/packages/playwright-crawler/src/internals/playwright-launcher.ts b/packages/playwright-crawler/src/internals/playwright-launcher.ts index 009c499c6bc9..63636d7e7908 100644 --- a/packages/playwright-crawler/src/internals/playwright-launcher.ts +++ b/packages/playwright-crawler/src/internals/playwright-launcher.ts @@ -83,9 +83,8 @@ export class PlaywrightLauncher extends BrowserLauncher { */ protected static override optionsShape = { ...BrowserLauncher.optionsShape, - // Passthrough schemas — the launcher module object must keep its prototype through parsing. + // Passthrough schema — the launcher module object must keep its prototype through parsing. launcher: schemas.anyObject.optional(), - launchContextOptions: schemas.anyObject.optional(), }; /** @internal */ diff --git a/packages/playwright-crawler/src/internals/utils/playwright-utils.ts b/packages/playwright-crawler/src/internals/utils/playwright-utils.ts index e6efa2626879..4183c70f9841 100644 --- a/packages/playwright-crawler/src/internals/utils/playwright-utils.ts +++ b/packages/playwright-crawler/src/internals/utils/playwright-utils.ts @@ -25,8 +25,8 @@ import vm from 'node:vm'; import { Configuration, KeyValueStore, type Request, serviceLocator, SessionError, validators } from '@crawlee/browser'; import type { BatchAddRequestsResult, Dictionary } from '@crawlee/types'; import type { CheerioAPI } from 'cheerio'; -import { expandShadowRoots, sleep } from '@crawlee/utils'; -import { parseArgument, schemas } from '@crawlee/utils/internal'; +import { sleep } from '@crawlee/utils'; +import { expandShadowRoots, parseArgument, schemas } from '@crawlee/utils/internal'; import type { Download, Page, Response, Route } from 'playwright'; import { z } from 'zod'; @@ -34,7 +34,6 @@ import { LruCache } from '@apify/datastructures'; import type { EnqueueLinksByClickingElementsOptions } from '../enqueue-links/click-elements.js'; import { enqueueLinksByClickingElements } from '../enqueue-links/click-elements.js'; -import { RenderingTypePredictor } from './rendering-type-prediction.js'; const getLog = () => serviceLocator.getChildLog('Playwright Utils'); @@ -674,7 +673,7 @@ export interface HandleCloudflareChallengeOptions { * @param url current URL for request identification, only used for logging * @param [options] */ -async function handleCloudflareChallenge( +export async function handleCloudflareChallenge( page: Page, url: string, options: HandleCloudflareChallengeOptions = {}, @@ -1024,18 +1023,3 @@ export interface PlaywrightContextUtils { } export { enqueueLinksByClickingElements }; - -/** @internal */ -export const playwrightUtils = { - injectFile, - injectJQuery, - gotoExtended, - blockRequests, - enqueueLinksByClickingElements, - parseWithCheerio, - infiniteScroll, - saveSnapshot, - compileScript, - RenderingTypePredictor, - handleCloudflareChallenge, -}; diff --git a/packages/playwright-crawler/src/internals/utils/rendering-type-prediction.ts b/packages/playwright-crawler/src/internals/utils/rendering-type-prediction.ts index 55f1972d510c..437b471ca089 100644 --- a/packages/playwright-crawler/src/internals/utils/rendering-type-prediction.ts +++ b/packages/playwright-crawler/src/internals/utils/rendering-type-prediction.ts @@ -128,7 +128,7 @@ const stateCodec = z.codec(persistedState, predictorState, { */ export class RenderingTypePredictor implements IRenderingTypePredictor { #detectionRatio: number; - #state: RecoverableState, z.input>; + readonly #state: RecoverableState, z.input>; constructor({ detectionRatio, persistenceOptions }: RenderingTypePredictorOptions) { this.#detectionRatio = detectionRatio; diff --git a/packages/puppeteer-crawler/src/index.ts b/packages/puppeteer-crawler/src/index.ts index 99c5a18ffe99..d75688df5c04 100644 --- a/packages/puppeteer-crawler/src/index.ts +++ b/packages/puppeteer-crawler/src/index.ts @@ -3,19 +3,9 @@ export * from './internals/puppeteer-browser-pool.js'; export * from './internals/puppeteer-crawler.js'; export * from './internals/puppeteer-launcher.js'; -export * as puppeteerRequestInterception from './internals/utils/puppeteer_request_interception.js'; export type { InterceptHandler } from './internals/utils/puppeteer_request_interception.js'; export * as puppeteerUtils from './internals/utils/puppeteer_utils.js'; -export type { - BlockRequestsOptions, - CompiledScriptFunction, - CompiledScriptParams, - DirectNavigationOptions as PuppeteerDirectNavigationOptions, - InfiniteScrollOptions, - InjectFileOptions, - SaveSnapshotOptions, -} from './internals/utils/puppeteer_utils.js'; +export type { DirectNavigationOptions as PuppeteerDirectNavigationOptions } from './internals/utils/puppeteer_utils.js'; -export * as puppeteerClickElements from './internals/enqueue-links/click-elements.js'; export type { EnqueueLinksByClickingElementsOptions } from './internals/enqueue-links/click-elements.js'; diff --git a/packages/puppeteer-crawler/src/internals/enqueue-links/click-elements.ts b/packages/puppeteer-crawler/src/internals/enqueue-links/click-elements.ts index f7372962ba30..886e9b0e789a 100644 --- a/packages/puppeteer-crawler/src/internals/enqueue-links/click-elements.ts +++ b/packages/puppeteer-crawler/src/internals/enqueue-links/click-elements.ts @@ -6,7 +6,6 @@ import type { RequestTransform, SkippedRequestCallback, UrlPatternInput, - UrlPatternObject, } from '@crawlee/browser'; import { applyRequestTransform, @@ -200,7 +199,7 @@ export interface EnqueueLinksByClickingElementsOptions { * **Example usage** * * ```javascript - * await utils.puppeteer.enqueueLinksByClickingElements({ + * await puppeteerUtils.enqueueLinksByClickingElements({ * page, * requestManager, * selector: 'a.product-detail', @@ -240,8 +239,8 @@ export async function enqueueLinksByClickingElements( const maxWaitForPageIdleMillis = maxWaitForPageIdleSecs * 1000; const hasOnSkippedRequest = onSkippedRequest !== undefined; - const urlExcludePatternObjects: UrlPatternObject[] = exclude?.length ? constructUrlPatternObjects(exclude) : []; - const urlPatternObjects: UrlPatternObject[] = include?.length ? constructUrlPatternObjects(include) : []; + const urlExcludePatternObjects = exclude?.length ? constructUrlPatternObjects(exclude) : []; + const urlPatternObjects = include?.length ? constructUrlPatternObjects(include) : []; const interceptedRequests = await clickElementsAndInterceptNavigationRequests({ page, @@ -300,6 +299,8 @@ interface ClickElementsAndInterceptNavigationRequestsOptions extends WaitForPage * Clicks all elements of given page matching given selector. * Catches and intercepts all initiated navigation requests and opened pages. * Returns a list of all target URLs. + * + * Not part of the public API — exported only so tests can import this module directly. * @ignore */ export async function clickElementsAndInterceptNavigationRequests( @@ -393,7 +394,7 @@ function createTargetCreatedHandler(page: Page, requests: Set): (target: * We're only interested in pages created by the page we're currently clicking in. * There will generally be a lot of other targets being created in the browser. */ -export function isTargetRelevant(page: Page, target: Target): boolean { +function isTargetRelevant(page: Page, target: Target): boolean { // oxlint-disable-next-line typescript/no-deprecated -- the non-deprecated replacement (opener.page()) is async and would force every call site to await, including EventEmitter callbacks return target.type() === 'page' && page.target() === target.opener(); } @@ -451,6 +452,8 @@ async function preventHistoryNavigation(page: Page): Promise { * so we first move them to the top of the page's stacking context and then click. * We do all in series to prevent elements from hiding one another. Therefore, * for large element sets, this will take considerable amount of time. + * + * Not part of the public API — exported only so tests can import this module directly. * @ignore */ export async function clickElements( diff --git a/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts b/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts index 7d6292823ee9..23228af0249d 100644 --- a/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts +++ b/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts @@ -10,12 +10,12 @@ import type { RouteSchemas, RoutesFromSchemas, } from '@crawlee/browser'; -import { assertBrowserPoolNotConfigured, BrowserCrawler, RequestState, Router } from '@crawlee/browser'; +import { BrowserCrawler, RequestState, Router } from '@crawlee/browser'; import { serviceLocator } from '@crawlee/core'; import type { Dictionary } from '@crawlee/types'; -import { parseArgument } from '@crawlee/utils/internal'; +import { assertBrowserPoolNotConfigured, parseArgument, schemas } from '@crawlee/utils/internal'; // @ts-ignore This only throws when compiled against puppeteer 25+ (ESM only), we only import types, so its alllll gooooood -import type { HTTPResponse, LaunchOptions, Page } from 'puppeteer'; +import type { HTTPResponse, Page } from 'puppeteer'; import { z } from 'zod'; import type { EnqueueLinksByClickingElementsOptions } from './enqueue-links/click-elements.js'; @@ -30,7 +30,7 @@ import type { PuppeteerContextUtils, SaveSnapshotOptions, } from './utils/puppeteer_utils.js'; -import { gotoExtended, puppeteerUtils } from './utils/puppeteer_utils.js'; +import * as puppeteerUtils from './utils/puppeteer_utils.js'; export type PuppeteerGoToOptions = NonNullable[1]>; @@ -189,7 +189,6 @@ export class PuppeteerCrawler< > extends BrowserCrawler< Page, HTTPResponse, - LaunchOptions, PuppeteerCrawlingContext, ContextExtension, ExtendedContext, @@ -201,6 +200,7 @@ export class PuppeteerCrawler< */ protected static override optionsShape = { ...BrowserCrawler.optionsShape, + launchContext: schemas.anyObject.default(() => ({})), // Deliberately looser than the declared type: Puppeteer's own accepted string values have moved over // time (`'new'`/`'old'`, now `'shell'`), and the value is forwarded to it verbatim. headless: z.union([z.boolean(), z.string()]).optional(), @@ -251,7 +251,6 @@ export class PuppeteerCrawler< Routes, StatisticStateExtension >), - launchContext, configuration, proxyConfiguration, browserPoolBuilder: (remoteBrowser) => @@ -321,7 +320,7 @@ export class PuppeteerCrawler< crawlingContext: PuppeteerCrawlingContext, gotoOptions: DirectNavigationOptions, ) { - return gotoExtended(crawlingContext.page, crawlingContext.request, gotoOptions); + return puppeteerUtils.gotoExtended(crawlingContext.page, crawlingContext.request, gotoOptions); } } diff --git a/packages/puppeteer-crawler/src/internals/utils/puppeteer_utils.ts b/packages/puppeteer-crawler/src/internals/utils/puppeteer_utils.ts index 10a8975fa52b..49c0e8d6bc53 100644 --- a/packages/puppeteer-crawler/src/internals/utils/puppeteer_utils.ts +++ b/packages/puppeteer-crawler/src/internals/utils/puppeteer_utils.ts @@ -5,7 +5,7 @@ * **Example usage:** * * ```javascript - * import { launchPuppeteer, utils } from 'crawlee'; + * import { launchPuppeteer, puppeteerUtils } from 'crawlee'; * * // Open https://www.example.com in Puppeteer * const browser = await launchPuppeteer(); @@ -13,7 +13,7 @@ * await page.goto('https://www.example.com'); * * // Inject jQuery into a page - * await utils.puppeteer.injectJQuery(page); + * await puppeteerUtils.injectJQuery(page); * ``` * @module puppeteerUtils */ @@ -26,11 +26,11 @@ import type { Request } from '@crawlee/browser'; import { Configuration, KeyValueStore, serviceLocator, validators } from '@crawlee/browser'; import type { BatchAddRequestsResult, Dictionary } from '@crawlee/types'; import type { CheerioAPI } from 'cheerio'; -import { expandShadowRoots, sleep } from '@crawlee/utils'; -import { parseArgument, schemas } from '@crawlee/utils/internal'; +import { sleep } from '@crawlee/utils'; +import { expandShadowRoots, parseArgument, schemas } from '@crawlee/utils/internal'; import type { ProtocolMapping } from 'devtools-protocol/types/protocol-mapping.js'; // @ts-ignore This only throws when compiled against puppeteer 25+ (ESM only), we only import types, so its alllll gooooood -import type { HTTPRequest as PuppeteerRequest, HTTPResponse, Page, ResponseForRequest } from 'puppeteer'; +import type { HTTPRequest as PuppeteerRequest, HTTPResponse, Page } from 'puppeteer'; import { z } from 'zod'; import { LruCache } from '@apify/datastructures'; @@ -50,7 +50,6 @@ const filePathSchema = z.string(); const injectFileOptionsSchema = z.strictObject({ surviveNavigations: z.boolean().optional(), }); -const responseUrlRulesSchema = schemas.arrayOf(z.union([z.string(), z.instanceof(RegExp)]), 'strings or RegExps'); const gotoExtendedRequestSchema = z.looseObject({ url: z.url(), method: z.string().optional(), @@ -187,7 +186,7 @@ export async function injectFile(page: Page, filePath: string, options: InjectFi * * **Example usage:** * ```javascript - * await utils.puppeteer.injectJQuery(page); + * await puppeteerUtils.injectJQuery(page); * const title = await page.evaluate(() => { * return $('head title').text(); * }); @@ -210,7 +209,7 @@ export async function injectJQuery(page: Page, options?: { surviveNavigations?: * * **Example usage:** * ```javascript - * const $ = await utils.puppeteer.parseWithCheerio(page); + * const $ = await puppeteerUtils.parseWithCheerio(page); * const title = $('title').text(); * ``` * @@ -294,13 +293,13 @@ export async function parseWithCheerio( * * **Example usage** * ```javascript - * import { launchPuppeteer, utils } from 'crawlee'; + * import { launchPuppeteer, puppeteerUtils } from 'crawlee'; * * const browser = await launchPuppeteer(); * const page = await browser.newPage(); * * // Block all requests to URLs that include `adsbygoogle.js` and also all defaults. - * await utils.puppeteer.blockRequests(page, { + * await puppeteerUtils.blockRequests(page, { * extraUrlPatterns: ['adsbygoogle.js'], * }); * @@ -323,7 +322,7 @@ export async function blockRequests(page: Page, options: BlockRequestsOptions = /** * @internal */ -export async function sendCDPCommand( +async function sendCDPCommand( page: Page, command: T, ...args: ProtocolMapping.Commands[T]['paramsType'] @@ -349,95 +348,6 @@ export async function sendCDPCommand( ); } -/** - * `blockResources()` has a high impact on performance in recent versions of Puppeteer. - * Until this resolves, please use `utils.puppeteer.blockRequests()`. - * @deprecated - */ -export const blockResources = async (page: Page, resourceTypes = ['stylesheet', 'font', 'image', 'media']) => { - serviceLocator - .getLogger() - .deprecated( - 'utils.puppeteer.blockResources() has a high impact on performance in recent versions of Puppeteer. ' + - 'Until this resolves, please use utils.puppeteer.blockRequests()', - ); - await addInterceptRequestHandler(page, async (request) => { - const type = request.resourceType(); - if (resourceTypes.includes(type)) await request.abort(); - else await request.continue(); - }); -}; - -/** - * *NOTE:* In recent versions of Puppeteer using this function entirely disables browser cache which resolves in sub-optimal - * performance. Until this resolves, we suggest just relying on the in-browser cache unless absolutely necessary. - * - * Enables caching of intercepted responses into a provided object. Automatically enables request interception in Puppeteer. - * *IMPORTANT*: Caching responses stores them to memory, so too loose rules could cause memory leaks for longer running crawlers. - * This issue should be resolved or atleast mitigated in future iterations of this feature. - * @param page - * Puppeteer [`Page`](https://pptr.dev/api/puppeteer.page) object. - * @param cache - * Object in which responses are stored - * @param responseUrlRules - * List of rules that are used to check if the response should be cached. - * String rules are compared as page.url().includes(rule) while RegExp rules are evaluated as rule.test(page.url()). - * @deprecated - */ -export async function cacheResponses( - page: Page, - cache: Dictionary>, - responseUrlRules: (string | RegExp)[], -): Promise { - parseArgument(page, validators.browserPage); - parseArgument(cache, schemas.anyObject); - parseArgument(responseUrlRules, responseUrlRulesSchema); - - serviceLocator - .getLogger() - .deprecated( - 'utils.puppeteer.cacheResponses() has a high impact on performance ' + - "in recent versions of Puppeteer so it's use is discouraged until this issue resolves.", - ); - - await addInterceptRequestHandler(page, async (request) => { - const url = request.url(); - - if (cache[url]) { - await request.respond(cache[url]); - return; - } - - await request.continue(); - }); - - page.on('response', async (response) => { - const url = response.url(); - - // Response is already cached, do nothing - if (cache[url]) return; - - const shouldCache = responseUrlRules.some((rule) => { - if (typeof rule === 'string') return url.includes(rule); - if (rule instanceof RegExp) return rule.test(url); - return false; - }); - - try { - if (shouldCache) { - const buffer = await response.buffer(); - cache[url] = { - status: response.status(), - headers: response.headers(), - body: buffer, - }; - } - } catch { - // ignore errors, usually means that buffer is empty or broken connection - } - }); -} - /** * Compiles a Puppeteer script into an async function that may be executed at any time * by providing it with the following object: @@ -1026,18 +936,3 @@ export interface PuppeteerContextUtils { } export { enqueueLinksByClickingElements, addInterceptRequestHandler, removeInterceptRequestHandler }; - -/** @internal */ -export const puppeteerUtils = { - injectFile, - injectJQuery, - enqueueLinksByClickingElements, - blockRequests, - compileScript, - gotoExtended, - addInterceptRequestHandler, - removeInterceptRequestHandler, - infiniteScroll, - saveSnapshot, - parseWithCheerio, -}; diff --git a/packages/stagehand-crawler/src/index.ts b/packages/stagehand-crawler/src/index.ts index 184441dbd1f0..71d92b4d07ef 100644 --- a/packages/stagehand-crawler/src/index.ts +++ b/packages/stagehand-crawler/src/index.ts @@ -70,23 +70,18 @@ export type { StagehandPage, StagehandCrawlingContext, StagehandHook, - StagehandRequestHandler, StagehandGotoOptions, StagehandCrawlerOptions, } from './internals/stagehand-crawler'; export type { StagehandLaunchContext } from './internals/stagehand-launcher'; -// Export utilities as namespace -export * as stagehandUtils from './internals/utils/stagehand-utils'; - // Re-export key types from Stagehand for convenience export type { ActOptions, ActResult, Action, AgentConfig, - AgentResult, ExtractOptions, ModelConfiguration, ObserveOptions, diff --git a/packages/stagehand-crawler/src/internals/stagehand-crawler.ts b/packages/stagehand-crawler/src/internals/stagehand-crawler.ts index 150d67ddc21c..1f438be4b953 100644 --- a/packages/stagehand-crawler/src/internals/stagehand-crawler.ts +++ b/packages/stagehand-crawler/src/internals/stagehand-crawler.ts @@ -18,7 +18,6 @@ import type { ContextPipeline, CrawlingContext, GetUserDataFromRequest, - LoadedContext, OwnedBrowserPool, RequestHandler, RouterHandler, @@ -26,10 +25,10 @@ import type { RouteSchemas, RoutesFromSchemas, } from '@crawlee/browser'; -import { assertBrowserPoolNotConfigured, BrowserCrawler, Router } from '@crawlee/browser'; +import { BrowserCrawler, Router } from '@crawlee/browser'; import type { Dictionary } from '@crawlee/types'; -import { parseArgument, schemas } from '@crawlee/utils/internal'; -import type { LaunchOptions, Page, Response } from 'playwright'; +import { assertBrowserPoolNotConfigured, parseArgument, schemas } from '@crawlee/utils/internal'; +import type { Page, Response } from 'playwright'; import { z } from 'zod'; import { remoteStagehandBrowserPool, stagehandBrowserPool } from './stagehand-browser-pool'; @@ -237,11 +236,6 @@ export type StagehandHook< UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler > = BrowserHook>; -/** - * Request handler for StagehandCrawler. - */ -export interface StagehandRequestHandler extends RequestHandler> {} - /** * Options for StagehandCrawler. */ @@ -324,9 +318,8 @@ export interface StagehandCrawlerOptions< * ``` */ // Both union members must share the exact same call signature, otherwise TS cannot contextually type - // an inline `requestHandler({ page, request, ... })`. `StagehandRequestHandler` wraps the context in - // `LoadedContext`, while the router member uses `ExtendedContext`; since `StagehandCrawlingContext` - // already carries a `LoadedRequest`, using `ExtendedContext` for both keeps the signatures identical. + // an inline `requestHandler({ page, request, ... })`. `StagehandCrawlingContext` already carries a + // `LoadedRequest`, so using `ExtendedContext` for both keeps the signatures identical. requestHandler?: RouterHandler | RequestHandler; /** @@ -406,7 +399,6 @@ export class StagehandCrawler< > extends BrowserCrawler< StagehandPage, Response, - LaunchOptions, StagehandCrawlingContext, ContextExtension, ExtendedContext, @@ -418,6 +410,7 @@ export class StagehandCrawler< */ protected static override optionsShape = { ...BrowserCrawler.optionsShape, + launchContext: schemas.anyObject.default(() => ({})), stagehandOptions: schemas.anyObject.optional(), headless: z.boolean().optional(), }; @@ -460,7 +453,6 @@ export class StagehandCrawler< Routes, StatisticStateExtension >), - launchContext, configuration, // The pool serves plain Playwright pages - a page only becomes a `StagehandPage` further down the // pipeline, once `setUpStagehand` enhances it - so its page type is narrower than the crawler's. diff --git a/packages/stagehand-crawler/src/internals/stagehand-launcher.ts b/packages/stagehand-crawler/src/internals/stagehand-launcher.ts index 7f069ef571ad..9c4d2e29839e 100644 --- a/packages/stagehand-crawler/src/internals/stagehand-launcher.ts +++ b/packages/stagehand-crawler/src/internals/stagehand-launcher.ts @@ -17,11 +17,6 @@ export interface StagehandLaunchContext extends BrowserLaunchContext[1]; - /** - * Stagehand-specific configuration for AI operations. - */ - stagehandOptions?: StagehandOptions; - /** * URL to a HTTP proxy server. It must define the port number, * and it may also contain proxy username and password. @@ -37,24 +32,6 @@ export interface StagehandLaunchContext extends BrowserLaunchContext { * All StagehandLauncher parameters are passed via the launchContext object. */ constructor( - launchContext: StagehandLaunchContext = {}, + // `stagehandOptions` is not part of the public `StagehandLaunchContext`: it is how + // `stagehandBrowserPool()` threads the crawler's resolved Stagehand options through the launcher. + launchContext: StagehandLaunchContext & { stagehandOptions?: StagehandOptions } = {}, override readonly configuration = Configuration.getGlobalConfiguration(), ) { const parsedContext = parseArgument(launchContext, StagehandLauncher.optionsSchema, 'StagehandLaunchContext'); diff --git a/packages/stagehand-crawler/src/internals/stagehand-plugin.ts b/packages/stagehand-crawler/src/internals/stagehand-plugin.ts index f74e4032257e..23e0b0e24389 100644 --- a/packages/stagehand-crawler/src/internals/stagehand-plugin.ts +++ b/packages/stagehand-crawler/src/internals/stagehand-plugin.ts @@ -34,12 +34,12 @@ export interface StagehandPluginOptions extends BrowserPluginOptions { - readonly stagehandOptions: StagehandOptions; + readonly #stagehandOptions: StagehandOptions; readonly #stagehandInstances: WeakMap = new WeakMap(); constructor(library: BrowserType, options: StagehandPluginOptions = {}) { super(library, options); - this.stagehandOptions = options.stagehandOptions ?? {}; + this.#stagehandOptions = options.stagehandOptions ?? {}; } /** @@ -51,7 +51,7 @@ export class StagehandPlugin extends BrowserPlugin): Error { const message = error instanceof Error ? error.message : String(error); - const model = this.stagehandOptions.model; + const model = this.#stagehandOptions.model; let helpText = ''; @@ -200,11 +200,4 @@ export class StagehandPlugin extends BrowserPlugin; -} - /** * A snapshot of the relevant state of a page, as extracted by * {@apilink IBrowserPool.extractPageState}. diff --git a/packages/types/src/http-client.ts b/packages/types/src/http-client.ts index f29a68a857b8..37221369bd29 100644 --- a/packages/types/src/http-client.ts +++ b/packages/types/src/http-client.ts @@ -20,20 +20,10 @@ export interface HttpRequest { cookieJar?: CookieJar; followRedirect?: boolean | ((response: any) => boolean); // TODO BC with got - specify type better in 4.0 - maxRedirects?: number; encoding?: BufferEncoding; - throwHttpErrors?: boolean; - // from got-scraping Context proxyUrl?: string; - headerGeneratorOptions?: Record; - useHeaderGenerator?: boolean; - headerGenerator?: { - getHeaders: (options: Record) => Record; - }; - insecureHTTPParser?: boolean; - sessionToken?: object; } /** @@ -54,14 +44,6 @@ export interface HttpRequestOptions extends HttpRequest { password?: string; } -/** - * Type of a function called when an HTTP redirect takes place. It is allowed to mutate the `updatedRequest` argument. - */ -export type RedirectHandler = ( - redirectResponse: Response, - updatedRequest: { url?: string | URL; headers: Headers }, -) => void; - export interface SendRequestOptions { session?: ISession; cookieJar?: CookieJar; @@ -84,10 +66,6 @@ export interface SendRequestOptions { ignoreTlsErrors?: boolean; } -export interface StreamOptions extends SendRequestOptions { - onRedirect?: RedirectHandler; -} - /** * Interface for user-defined HTTP clients to be used for plain HTTP crawling and for sending additional requests during a crawl. */ diff --git a/packages/utils/src/index.ts b/packages/utils/src/index.ts index 37615e56b893..648de99335d6 100644 --- a/packages/utils/src/index.ts +++ b/packages/utils/src/index.ts @@ -2,10 +2,11 @@ export { htmlToText } from './internals/cheerio.js'; export { downloadListOfUrls, extractUrls } from './internals/extract-urls.js'; export { EnqueueStrategy } from './internals/url.js'; export type { DownloadListOfUrlsOptions, ExtractUrlsOptions } from './internals/extract-urls.js'; -export { sleep, expandShadowRoots } from './internals/general.js'; +export { sleep } from './internals/general.js'; export * as social from './internals/social.js'; export * from './internals/extract-microdata.js'; export * from './internals/open_graph_parser.js'; export * from './internals/robots.js'; -export * from './internals/sitemap.js'; +export { discoverValidSitemaps, Sitemap } from './internals/sitemap.js'; +export type { ParseSitemapOptions } from './internals/sitemap.js'; export { ArgumentValidationError } from './internals/validation.js'; diff --git a/packages/utils/src/internal.ts b/packages/utils/src/internal.ts index 6f4af65e0ad6..4582eb7296dd 100644 --- a/packages/utils/src/internal.ts +++ b/packages/utils/src/internal.ts @@ -2,8 +2,9 @@ export * from './internals/blocked.js'; export type { CheerioAPI, Cheerio, Element } from './internals/cheerio.js'; export { extractUrlsFromCheerio } from './internals/cheerio.js'; export { tryAbsoluteURL } from './internals/extract-urls.js'; -export { URL_NO_COMMAS_REGEX, URL_WITH_COMMAS_REGEX } from './internals/general.js'; +export { expandShadowRoots, URL_NO_COMMAS_REGEX, URL_WITH_COMMAS_REGEX } from './internals/general.js'; export * from './internals/iterables.js'; export * from './internals/url.js'; export * from './internals/validation.js'; export * as schemas from './internals/schemas.js'; +export { parseSitemap } from './internals/sitemap.js'; diff --git a/packages/utils/src/internals/social.ts b/packages/utils/src/internals/social.ts index 9293934dcd85..9bec60c18190 100644 --- a/packages/utils/src/internals/social.ts +++ b/packages/utils/src/internals/social.ts @@ -386,7 +386,7 @@ export const TWITTER_REGEX = new RegExp(`^${TWITTER_REGEX_STRING}$`, 'i'); * ``` * import { social } from 'crawlee'; * - * const matches = text.match(social.TWITTER_REGEX_STRING); + * const matches = text.match(social.TWITTER_REGEX_GLOBAL); * if (matches) console.log(`${matches.length} Twitter profiles found!`); * ``` */ diff --git a/packages/utils/src/internals/validation.ts b/packages/utils/src/internals/validation.ts index ef06b77a636b..22eabfe68eea 100644 --- a/packages/utils/src/internals/validation.ts +++ b/packages/utils/src/internals/validation.ts @@ -1 +1,23 @@ +import type { Dictionary } from '@crawlee/types'; + export { ArgumentValidationError, parseArgument } from '@apify/validations'; + +/** + * Rejects options that exist only to configure the browser pool the crawler would have built for itself. + * Accepting them alongside a pre-built `browserPool` and quietly ignoring them is how `browserPoolOptions` grew + * into a second, half-working way of configuring the same pool. + * @internal + */ +export function assertBrowserPoolNotConfigured(crawlerName: string, ignoredOptions: Dictionary): void { + const names = Object.keys(ignoredOptions).filter((name) => ignoredOptions[name] !== undefined); + + if (names.length === 0) { + return; + } + + throw new Error( + `${crawlerName}: ${names.map((name) => `\`${name}\``).join(', ')} cannot be combined with \`browserPool\`, ` + + `${names.length > 1 ? 'they configure' : 'it configures'} the browser pool the crawler would build for ` + + 'itself. Configure the pool you pass in instead.', + ); +} diff --git a/test/browser-pool/browser-pool.test.ts b/test/browser-pool/browser-pool.test.ts index af74541fb427..f4026514c807 100644 --- a/test/browser-pool/browser-pool.test.ts +++ b/test/browser-pool/browser-pool.test.ts @@ -93,12 +93,18 @@ describe.each([ describe('Initialization & retirement', () => { test('should retire browsers', async () => { - await browserPool.newPage(); + // The pool's controller sets are private; retirement is observable through the event. + const retiredControllers: BrowserController[] = []; + browserPool.on(BROWSER_POOL_EVENTS.BROWSER_RETIRED, (controller) => { + retiredControllers.push(controller); + }); + + const page = await browserPool.newPage(); + const controller = browserPool.getBrowserControllerByPage(page)!; browserPool.retireAllBrowsers(); - expect(browserPool.startingBrowserControllers.size).toBe(0); - expect(browserPool.activeBrowserControllers.size).toBe(0); - expect(browserPool.retiredBrowserControllers.size).toBe(1); + + expect(retiredControllers).toEqual([controller]); }); test('should destroy pool', async () => { @@ -109,9 +115,6 @@ describe.each([ await browserPool.destroy(); expect(browserController.close).toHaveBeenCalled(); - expect(browserPool.startingBrowserControllers.size).toBe(0); - expect(browserPool.activeBrowserControllers.size).toBe(0); - expect(browserPool.retiredBrowserControllers.size).toBe(0); expect(browserPool['browserKillerInterval']).toBeUndefined(); }); }); @@ -127,8 +130,14 @@ describe.each([ // https://github.com/apify/crawlee/issues/3670 test('should not leak aborted cancelTask between concurrent newPage calls', async () => { - const previousTimeout = browserPool.operationTimeoutMillis; - browserPool.operationTimeoutMillis = 1; + // The pool's timeout mirror is private, so the tiny timeout is set at construction. + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + operationTimeoutSecs: 0.001, + }); // Each newPage call is wrapped in its own outer addTimeoutToPromise, // matching how BasicCrawler wraps _runRequestHandler. Without the fix, @@ -142,7 +151,6 @@ describe.each([ ), ); - browserPool.operationTimeoutMillis = previousTimeout; browserPool.retireAllBrowsers(); // All calls must reject — none should silently resolve with undefined. @@ -159,8 +167,13 @@ describe.each([ // TODO: this test is very flaky in the CI test.skip('should allow early aborting in case of outer timeout', async () => { - const timeout = browserPool.operationTimeoutMillis; - browserPool.operationTimeoutMillis = 500; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + operationTimeoutSecs: 0.5, + }); // @ts-expect-error mocking private method const spy = vitest.spyOn(BrowserPool.prototype, 'executeHooks'); @@ -178,7 +191,6 @@ describe.each([ // 4 calls instead of just one. expect(spy).toBeCalledTimes(1); - browserPool.operationTimeoutMillis = timeout; browserPool.retireAllBrowsers(); }); @@ -202,8 +214,6 @@ describe.each([ await browserPool.newPageInNewBrowser(); await browserPool.newPageInNewBrowser(); - expect(browserPool.startingBrowserControllers.size).toBe(0); - expect(browserPool.activeBrowserControllers.size).toBe(3); expect(plugin.launch).toHaveBeenCalledTimes(3); }); @@ -249,8 +259,8 @@ describe.each([ expect(pageClosed).toHaveBeenCalled(); // The page is still attached, so the browser cannot be trusted with more work. - expect(browserPool.activeBrowserControllers.has(controller)).toBe(false); - expect(browserPool.retiredBrowserControllers.has(controller)).toBe(true); + expect(browserPool['activeBrowserControllers'].has(controller)).toBe(false); + expect(browserPool['retiredBrowserControllers'].has(controller)).toBe(true); }); test('should not retire the browser when only a post-close hook hangs', async () => { @@ -291,8 +301,8 @@ describe.each([ expect(pool['pages'].has(pageId)).toBe(false); // Retirement is keyed on the page, not on the timeout, so a slow hook must not // cost a browser that closed its page just fine. - expect(pool.activeBrowserControllers.has(controller)).toBe(true); - expect(pool.retiredBrowserControllers.has(controller)).toBe(false); + expect(pool['activeBrowserControllers'].has(controller)).toBe(true); + expect(pool['retiredBrowserControllers'].has(controller)).toBe(false); } finally { releaseHook?.(); await pool.destroy(); @@ -359,7 +369,7 @@ describe.each([ expect(controller.activePages).toEqual(0); expect(browserPool['pages'].has(pageId)).toBe(false); - expect(browserPool.retiredBrowserControllers.has(controller)).toBe(true); + expect(browserPool['retiredBrowserControllers'].has(controller)).toBe(true); }, 30_000); test("should not cancel the caller's task when it gives up on a close", async () => { @@ -385,38 +395,49 @@ describe.each([ }, 45_000); test('should retire browser after page count', async () => { - browserPool.retireBrowserAfterPageCount = 2; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + retireBrowserAfterPageCount: 2, + }); vitest.spyOn(browserPool, 'retireBrowserController'); - expect(browserPool.activeBrowserControllers.size).toBe(0); await browserPool.newPage(); await browserPool.newPage(); await browserPool.newPage(); - expect(browserPool.activeBrowserControllers.size).toBe(1); - expect(browserPool.retiredBrowserControllers.size).toBe(1); - expect(browserPool.retireBrowserController).toBeCalledTimes(1); }); test('should allow max pages per browser', async () => { - browserPool.maxOpenPagesPerBrowser = 1; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + maxOpenPagesPerBrowser: 1, + }); // @ts-expect-error Private function vitest.spyOn(browserPool!, 'launchBrowser'); await browserPool.newPage(); - expect(browserPool.activeBrowserControllers.size).toBe(1); await browserPool.newPage(); - expect(browserPool.activeBrowserControllers.size).toBe(2); await browserPool.newPage(); - expect(browserPool.activeBrowserControllers.size).toBe(3); expect(browserPool['launchBrowser']).toBeCalledTimes(3); }); test('should allow max pages per browser - no race condition', async () => { - browserPool.maxOpenPagesPerBrowser = 1; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + maxOpenPagesPerBrowser: 1, + }); // @ts-expect-error Private function vitest.spyOn(browserPool, 'launchBrowser'); @@ -426,13 +447,22 @@ describe.each([ await Promise.all([browserPool.newPage(usePlugin), browserPool.newPage(usePlugin)]); - expect(browserPool.activeBrowserControllers.size).toBe(2); - expect(browserPool['launchBrowser']).toBeCalledTimes(2); }); test('should close retired browsers', async () => { - browserPool.retireBrowserAfterPageCount = 1; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + retireBrowserAfterPageCount: 1, + }); + + let retiredBrowsers = 0; + browserPool.on(BROWSER_POOL_EVENTS.BROWSER_RETIRED, () => { + retiredBrowsers++; + }); clearInterval(browserPool['browserKillerInterval']!); @@ -443,13 +473,13 @@ describe.each([ // @ts-expect-error Private function vitest.spyOn(browserPool!, 'closeRetiredBrowserWithNoPages'); - expect(browserPool.retiredBrowserControllers.size).toBe(0); + expect(retiredBrowsers).toBe(0); const page = await browserPool.newPage(); const controller = browserPool.getBrowserControllerByPage(page)!; vitest.spyOn(controller, 'close'); - expect(browserPool.retiredBrowserControllers.size).toBe(1); + expect(retiredBrowsers).toBe(1); await page.close(); await new Promise((resolve) => @@ -460,7 +490,6 @@ describe.each([ expect(browserPool['closeRetiredBrowserWithNoPages']).toHaveBeenCalled(); expect(controller.close).toHaveBeenCalled(); - expect(browserPool.retiredBrowserControllers.size).toBe(0); }); describe('hooks', () => { @@ -487,14 +516,6 @@ describe.each([ // times during the wait, instead of sleeping past the 2s defaults from beforeEach. // The waits stay real: a live browser is launching underneath, and faking the clock // would stall the driver's own timeouts along with the pool's. - const pool = new BrowserPool({ - browserPlugins: [plugin], - closeInactiveBrowserAfterSecs: 0.5, - retireInactiveBrowserAfterSecs: 0.5, - }); - clearInterval(pool['browserKillerInterval']!); - pool['browserKillerInterval'] = setInterval(async () => pool['closeInactiveRetiredBrowsers'](), 100); - let resolvePreLaunchHook: (() => void) | null = null; let resolvePostLaunchHook: (() => void) | null = null; @@ -505,18 +526,35 @@ describe.each([ resolvePostLaunchHook = resolve; }); - pool.preLaunchHooks = [...pool.preLaunchHooks, async () => preLaunchPromise]; + const pool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 0.5, + retireInactiveBrowserAfterSecs: 0.5, + preLaunchHooks: [async () => preLaunchPromise], + postLaunchHooks: [async () => postLaunchPromise], + }); + clearInterval(pool['browserKillerInterval']!); + pool['browserKillerInterval'] = setInterval(async () => pool['closeInactiveRetiredBrowsers'](), 100); - pool.postLaunchHooks = [...pool.postLaunchHooks, async () => postLaunchPromise]; + // The hook arrays are private, so they are supplied at construction above. Launch and + // retire bookkeeping is observed through the pool's events instead of its internals. + let launchedBrowsers = 0; + let retiredBrowsers = 0; + pool.on(BROWSER_POOL_EVENTS.BROWSER_LAUNCHED, () => { + launchedBrowsers++; + }); + pool.on(BROWSER_POOL_EVENTS.BROWSER_RETIRED, () => { + retiredBrowsers++; + }); try { const newPagePromise = pool.newPage(); await sleep(200); - expect(pool.startingBrowserControllers.size).toBe(1); - expect(pool.activeBrowserControllers.size).toBe(0); - expect(pool.retiredBrowserControllers.size).toBe(0); + // The browser is still starting - and it must not be retired while its hooks run. + expect(launchedBrowsers).toBe(0); + expect(retiredBrowsers).toBe(0); await sleep(1200); @@ -525,9 +563,8 @@ describe.each([ const page = await newPagePromise; - expect(pool.startingBrowserControllers.size).toBe(0); - expect(pool.activeBrowserControllers.size).toBe(1); - expect(pool.retiredBrowserControllers.size).toBe(0); + expect(launchedBrowsers).toBe(1); + expect(retiredBrowsers).toBe(0); // Make sure the page is usable. The Puppeteer and Playwright `evaluate` // overloads have no compatible signature, so the union is not callable as-is. @@ -541,7 +578,13 @@ describe.each([ describe('preLaunchHooks', () => { test('should evaluate hook before launching browser with correct args', async () => { const myAsyncHook = async () => Promise.resolve(); - browserPool.preLaunchHooks.push(myAsyncHook); + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + preLaunchHooks: [myAsyncHook], + }); // @ts-expect-error Private function vitest.spyOn(browserPool!, 'executeHooks'); @@ -551,7 +594,7 @@ describe.each([ const { launchContext } = browserPool.getBrowserControllerByPage(page)!; expect(browserPool['executeHooks']).toHaveBeenNthCalledWith( 1, - browserPool.preLaunchHooks, + expect.arrayContaining([myAsyncHook]), pageId, launchContext, ); @@ -562,11 +605,22 @@ describe.each([ // in limbo and subsequent newPage() calls would never resolve. test('error in hook does not leave browser stuck in limbo', async () => { const errorMessage = 'pre-launch failed'; - browserPool.preLaunchHooks = [ - async () => { - throw new Error(errorMessage); - }, - ]; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + preLaunchHooks: [ + async () => { + throw new Error(errorMessage); + }, + ], + }); + + let launchedBrowsers = 0; + browserPool.on(BROWSER_POOL_EVENTS.BROWSER_LAUNCHED, () => { + launchedBrowsers++; + }); const attempts = 5; for (let i = 0; i < attempts; i++) { @@ -577,7 +631,7 @@ describe.each([ } } - expect(browserPool.activeBrowserControllers.size).toBe(0); + expect(launchedBrowsers).toBe(0); expect.assertions(attempts + 1); }); }); @@ -585,7 +639,13 @@ describe.each([ describe('postLaunchHooks', () => { test('should evaluate hook after launching browser with correct args', async () => { const myAsyncHook = async () => Promise.resolve(); - browserPool.postLaunchHooks = [myAsyncHook]; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + postLaunchHooks: [myAsyncHook], + }); // @ts-expect-error Private function vitest.spyOn(browserPool, 'executeHooks'); @@ -596,7 +656,7 @@ describe.each([ expect(browserPool['executeHooks']).toHaveBeenNthCalledWith( 2, - browserPool.postLaunchHooks, + [myAsyncHook], pageId, browserController, ); @@ -608,12 +668,23 @@ describe.each([ test('error in hook does not leave browser stuck in limbo', async () => { const errorMessage = 'post-launch failed'; const controllers: BrowserController[] = []; - browserPool.postLaunchHooks = [ - async (_pageId, browserController) => { - controllers.push(browserController); - throw new Error(errorMessage); - }, - ]; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + postLaunchHooks: [ + async (_pageId, browserController) => { + controllers.push(browserController); + throw new Error(errorMessage); + }, + ], + }); + + let launchedBrowsers = 0; + browserPool.on(BROWSER_POOL_EVENTS.BROWSER_LAUNCHED, () => { + launchedBrowsers++; + }); const attempts = 5; for (let i = 0; i < attempts; i++) { @@ -636,7 +707,7 @@ describe.each([ }, 10); }); - expect(browserPool.activeBrowserControllers.size).toBe(0); + expect(launchedBrowsers).toBe(0); expect.assertions(attempts + 1); }); }); @@ -644,7 +715,13 @@ describe.each([ describe('prePageCreateHooks', () => { test('should evaluate hook after launching browser with correct args', async () => { const myAsyncHook = async () => Promise.resolve(); - browserPool.prePageCreateHooks = [myAsyncHook]; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + prePageCreateHooks: [myAsyncHook], + }); // @ts-expect-error Private function vitest.spyOn(browserPool, 'executeHooks'); @@ -655,7 +732,7 @@ describe.each([ expect(browserPool['executeHooks']).toHaveBeenNthCalledWith( 3, - browserPool.prePageCreateHooks, + expect.arrayContaining([myAsyncHook]), pageId, browserController, browserController.launchContext.useIncognitoPages ? {} : undefined, @@ -666,7 +743,13 @@ describe.each([ describe('postPageCreateHooks', () => { test('should evaluate hook after launching browser with correct args', async () => { const myAsyncHook = async () => Promise.resolve(); - browserPool.postPageCreateHooks = [myAsyncHook]; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + postPageCreateHooks: [myAsyncHook], + }); // @ts-expect-error Private function vitest.spyOn(browserPool, 'executeHooks'); @@ -676,7 +759,7 @@ describe.each([ expect(browserPool['executeHooks']).toHaveBeenNthCalledWith( 4, - browserPool.postPageCreateHooks, + expect.arrayContaining([myAsyncHook]), page, browserController, ); @@ -686,7 +769,13 @@ describe.each([ describe('prePageCloseHooks', () => { test('should evaluate hook after launching browser with correct args', async () => { const myAsyncHook = async () => Promise.resolve(); - browserPool.prePageCloseHooks = [myAsyncHook]; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + prePageCloseHooks: [myAsyncHook], + }); // @ts-expect-error Private function vitest.spyOn(browserPool, 'executeHooks'); @@ -697,7 +786,7 @@ describe.each([ const browserController = browserPool.getBrowserControllerByPage(page); expect(browserPool['executeHooks']).toHaveBeenNthCalledWith( 5, - browserPool.prePageCloseHooks, + [myAsyncHook], page, browserController, ); @@ -707,7 +796,13 @@ describe.each([ describe('postPageCloseHooks', () => { test('should evaluate hook after launching browser with correct args', async () => { const myAsyncHook = async () => Promise.resolve(); - browserPool.postPageCloseHooks = [myAsyncHook]; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + postPageCloseHooks: [myAsyncHook], + }); // @ts-expect-error Private function vitest.spyOn(browserPool, 'executeHooks'); @@ -719,7 +814,7 @@ describe.each([ const browserController = browserPool.getBrowserControllerByPage(page); expect(browserPool['executeHooks']).toHaveBeenNthCalledWith( 6, - browserPool.postPageCloseHooks, + [myAsyncHook], pageId, browserController, ); @@ -729,7 +824,14 @@ describe.each([ describe('events', () => { test(`should emit ${BROWSER_POOL_EVENTS.BROWSER_LAUNCHED} event`, async () => { - browserPool.maxOpenPagesPerBrowser = 1; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + maxOpenPagesPerBrowser: 1, + }); + let calls = 0; let argument; @@ -745,7 +847,14 @@ describe.each([ }); test(`should emit ${BROWSER_POOL_EVENTS.BROWSER_RETIRED} event`, async () => { - browserPool.retireBrowserAfterPageCount = 1; + await browserPool.destroy(); + browserPool = new BrowserPool({ + browserPlugins: [plugin], + closeInactiveBrowserAfterSecs: 2, + retireInactiveBrowserAfterSecs: 2, + retireBrowserAfterPageCount: 1, + }); + let calls = 0; let argument; browserPool.on(BROWSER_POOL_EVENTS.BROWSER_RETIRED, (arg) => { diff --git a/test/core/autoscaling/autoscaled_pool.test.ts b/test/core/autoscaling/autoscaled_pool.test.ts index 0ec2540e7033..f5ddff86f0e9 100644 --- a/test/core/autoscaling/autoscaled_pool.test.ts +++ b/test/core/autoscaling/autoscaled_pool.test.ts @@ -150,13 +150,13 @@ describe('AutoscaledPool', () => { const promise = pool.run(); - // Concurrency is tuned on the governor; the pool only reflects it as read-only telemetry. + // Concurrency is tuned on the governor; the pool only reflects it as read-only telemetry. `desiredConcurrency` + // is autoscaler-owned, so it is retuned indirectly - by moving the bounds it is clamped into. system.minConcurrency = 4; - system.maxConcurrency = 14; - system.desiredConcurrency = 7; + system.maxConcurrency = 7; expect(system.minConcurrency).toBe(4); - expect(system.maxConcurrency).toBe(14); + expect(system.maxConcurrency).toBe(7); expect(pool.desiredConcurrency).toBe(7); await promise; @@ -224,9 +224,20 @@ describe('AutoscaledPool', () => { expect(pool.desiredConcurrency).toBe(3); }); - test('works with high values', () => { + test('works with high values', async () => { + // A starting budget of 50 has to be configured, since `desiredConcurrency` is autoscaler-owned. + pool = await makePool( + { + runTaskFunction: async () => {}, + isFinishedFunction: async () => false, + isTaskReadyFunction: async () => true, + }, + { minConcurrency: 1, maxConcurrency: 100, desiredConcurrency: 50 }, + ); + // @ts-expect-error Mock + pool.system.systemStatus = systemStatus; + // Should not scale because current concurrency is too low. - systemOf(pool).desiredConcurrency = 50; const targetConcurrency = Math.floor( // @ts-expect-error Accessing private prop on the governor pool.desiredConcurrency * pool.system.desiredConcurrencyRatio, @@ -273,12 +284,11 @@ describe('AutoscaledPool', () => { isFinishedFunction: async () => count >= limit, isTaskReadyFunction: async () => count < limit, }, - { minConcurrency: 1, maxConcurrency: 100 }, + { minConcurrency: 1, maxConcurrency: 100, desiredConcurrency: 10 }, ); // @ts-expect-error Mock pool.system.systemStatus = systemStatus; systemStatus.okNow = false; - systemOf(pool).desiredConcurrency = 10; const origStart = pool.system.tryRegisterTaskStart.bind(pool.system); const origEnd = pool.system.registerTaskEnd.bind(pool.system); diff --git a/test/core/autoscaling/concurrency_system.test.ts b/test/core/autoscaling/concurrency_system.test.ts index 5dc301a3376b..bbfdb32b4671 100644 --- a/test/core/autoscaling/concurrency_system.test.ts +++ b/test/core/autoscaling/concurrency_system.test.ts @@ -94,16 +94,6 @@ describe('ConcurrencySystem', () => { expect(system.desiredConcurrency).toBe(20); }); - test('desiredConcurrency is held within the current bounds', () => { - const system = new ConcurrencySystem({ minConcurrency: 5, maxConcurrency: 50 }); - - system.desiredConcurrency = 500; - expect(system.desiredConcurrency).toBe(50); - - system.desiredConcurrency = 1; - expect(system.desiredConcurrency).toBe(5); - }); - test('the maximum wins a contradictory pair of bounds', () => { const system = new ConcurrencySystem({ minConcurrency: 10, maxConcurrency: 100, desiredConcurrency: 50 }); @@ -701,8 +691,9 @@ describe('ConcurrencySystem', () => { const first = makeBorrower('first'); const second = makeBorrower('second'); - // Tuning happens on the governor its owner holds; the pools only report it, read-only. - system.desiredConcurrency = 42; + // Tuning happens on the governor its owner holds; the pools only report it, read-only. `desiredConcurrency` + // is autoscaler-owned, so it is retuned by moving the bounds it is clamped into. + system.minConcurrency = 42; expect(first.desiredConcurrency).toBe(42); expect(second.desiredConcurrency).toBe(42); }); diff --git a/test/core/browser_launchers/playwright_launcher.test.ts b/test/core/browser_launchers/playwright_launcher.test.ts index 374edac79e67..9c4c81c54e94 100644 --- a/test/core/browser_launchers/playwright_launcher.test.ts +++ b/test/core/browser_launchers/playwright_launcher.test.ts @@ -5,13 +5,10 @@ import type { AddressInfo } from 'node:net'; import path from 'node:path'; import util from 'node:util'; -import { - BrowserLauncher, - Configuration, - launchPlaywright, - PlaywrightLauncher, - serviceLocator, -} from '@crawlee/playwright'; +import { BrowserLauncher, Configuration, launchPlaywright, serviceLocator } from '@crawlee/playwright'; +// `PlaywrightLauncher` is intentionally not part of the package's public exports; these tests need the +// class itself to inspect `createBrowserPlugin()` without launching a browser. +import { PlaywrightLauncher } from '../../../packages/playwright-crawler/src/internals/playwright-launcher.js'; // @ts-expect-error no types import basicAuthParser from 'basic-auth-parser'; import type { Browser, BrowserType } from 'playwright'; diff --git a/test/core/crawlers/basic_browser_crawler.ts b/test/core/crawlers/basic_browser_crawler.ts index 41ddf3827dbf..4d5dbdd9df64 100644 --- a/test/core/crawlers/basic_browser_crawler.ts +++ b/test/core/crawlers/basic_browser_crawler.ts @@ -9,7 +9,7 @@ import type { import { BrowserCrawler } from '@crawlee/puppeteer'; import type { Dictionary } from '@crawlee/types'; // @ts-ignore This only throws when compiled against puppeteer 25+ (ESM only), we only import types, so its alllll gooooood -import type { HTTPResponse, LaunchOptions, Page } from 'puppeteer'; +import type { HTTPResponse, Page } from 'puppeteer'; export type TestCrawlingContext = BrowserCrawlingContext; @@ -20,7 +20,7 @@ type TestBrowserPoolOptions = BrowserPoolOptions & Page >; -export class BrowserCrawlerTest extends BrowserCrawler { +export class BrowserCrawlerTest extends BrowserCrawler { constructor( options: Partial> & { /** diff --git a/test/core/crawlers/basic_crawler.test.ts b/test/core/crawlers/basic_crawler.test.ts index 2c9fc6c89655..9113a1ccee84 100644 --- a/test/core/crawlers/basic_crawler.test.ts +++ b/test/core/crawlers/basic_crawler.test.ts @@ -51,11 +51,12 @@ import { afterAll, beforeAll, beforeEach, describe, expect, test, vitest } from import { z } from 'zod'; import { startExpressAppPromise } from '../../shared/_helper.js'; +// `createRequestQueueBackend` is typed with the `@crawlee/types` interface, so the memory-specific +// `listItems()` helper these tests use has to come from the implementation class itself. +import type { RequestQueueBackend as MemoryRequestQueueBackend } from '../../../packages/core/src/memory-storage/resource-clients/request-queue.js'; import log from '@apify/log'; -type MemoryRequestQueueBackend = Awaited>; - describe('BasicCrawler', () => { let logLevel: number; let requestQueueBackend: MemoryRequestQueueBackend; @@ -363,8 +364,9 @@ describe('BasicCrawler', () => { await crawler.run(['https://example.com/1']); const firstSystem = crawler.concurrencySystem! as ConcurrencySystem; - // Simulate scaling state left behind by the first run. - firstSystem.desiredConcurrency = 42; + // Simulate scaling state left behind by the first run - `desiredConcurrency` is autoscaler-owned, so it is + // pushed up indirectly, by raising the floor it is clamped against. + firstSystem.minConcurrency = 42; await crawler.run(['https://example.com/2']); const secondSystem = crawler.concurrencySystem!; @@ -3708,6 +3710,7 @@ describe('BasicCrawler', () => { test('afterStorageCommit turns a rejected write into a non-retryable request failure', async () => { const dataset = await Dataset.open(); vitest + // @ts-expect-error Accessing private property .spyOn(dataset.backend, 'pushData') .mockRejectedValue(new Error('Data item is too large (size: 10000000 bytes)')); diff --git a/test/core/crawlers/browser_crawler.test.ts b/test/core/crawlers/browser_crawler.test.ts index 9234229856a0..2f162546d27d 100644 --- a/test/core/crawlers/browser_crawler.test.ts +++ b/test/core/crawlers/browser_crawler.test.ts @@ -1028,6 +1028,11 @@ describe('BrowserCrawler', () => { browserPlugins: [puppeteerPlugin], maxOpenPagesPerBrowser: 1, retireBrowserAfterPageCount: 1, + postLaunchHooks: [ + (_pageId, browserController) => { + browserProxies.push((browserController as PuppeteerController).launchContext.proxyUrl!); + }, + ], }, requestList, requestHandler: async () => {}, @@ -1036,10 +1041,6 @@ describe('BrowserCrawler', () => { maxConcurrency: 1, }); - (browserCrawler.browserPool as BrowserPool).postLaunchHooks.push((_pageId, browserController) => { - browserProxies.push((browserController as PuppeteerController).launchContext.proxyUrl!); - }); - await browserCrawler.run(); for (const proxyUrl of proxyUrls) { diff --git a/test/core/crawlers/playwright_crawler.test.ts b/test/core/crawlers/playwright_crawler.test.ts index b6a8be130919..5cddb476eca5 100644 --- a/test/core/crawlers/playwright_crawler.test.ts +++ b/test/core/crawlers/playwright_crawler.test.ts @@ -204,12 +204,13 @@ describe('PlaywrightCrawler', () => { describe('playwrightBrowserPool', () => { test('runs a plugin for the requested browser and forwards the pool options', () => { const browserPool = playwrightBrowserPool({ - maxOpenPagesPerBrowser: 3, + useFingerprints: false, headless: false, launchContext: { launcher: playwright.firefox }, }); - expect(browserPool.maxOpenPagesPerBrowser).toBe(3); + // The pool keeps its options private; `useFingerprints` is observable through the generator. + expect(browserPool.fingerprintGenerator).toBeUndefined(); expect(browserPool.browserPlugins).toHaveLength(1); expect(browserPool.browserPlugins[0].library).toBe(playwright.firefox); expect(browserPool.browserPlugins[0].launchOptions).toMatchObject({ headless: false }); @@ -217,8 +218,9 @@ describe('PlaywrightCrawler', () => { test('turns off fingerprint injection when a custom userAgent is given', () => { expect( - playwrightBrowserPool({ launchContext: { userAgent: 'Definitely Not A Crawler' } }).useFingerprints, - ).toBe(false); + playwrightBrowserPool({ launchContext: { userAgent: 'Definitely Not A Crawler' } }) + .fingerprintGenerator, + ).toBeUndefined(); }); test('is rejected by the crawler alongside the options that would configure its own pool', () => { diff --git a/test/core/crawlers/rendering_type_predictor.test.ts b/test/core/crawlers/rendering_type_predictor.test.ts index 82f3ba1c5cbc..e068a0c851d8 100644 --- a/test/core/crawlers/rendering_type_predictor.test.ts +++ b/test/core/crawlers/rendering_type_predictor.test.ts @@ -25,7 +25,7 @@ describe('RenderingTypePredictor', () => { predictor.storeResult(staticRequest, 'static'); predictor.storeResult(clientRequest, 'clientOnly'); - // Persist the state + // Persist the state - `teardown()` flushes it, and the predictor is not used again here. const store = await KeyValueStore.open(); await predictor.teardown(); diff --git a/test/core/enqueue_links/click_elements.test.ts b/test/core/enqueue_links/click_elements.test.ts index 6b4d6a952177..5c82a4776c1f 100644 --- a/test/core/enqueue_links/click_elements.test.ts +++ b/test/core/enqueue_links/click_elements.test.ts @@ -7,14 +7,16 @@ import { MemoryStorageBackend, playwrightClickElements, playwrightUtils, - puppeteerClickElements, puppeteerUtils, RequestQueue, serviceLocator, } from 'crawlee'; import type { Browser as PWBrowser, Page as PWPage } from 'playwright'; // @ts-ignore This only throws when compiled against puppeteer 25+ (ESM only), we only import types, so its alllll gooooood -import type { Browser as PPBrowser, Target } from 'puppeteer'; +import type { Browser as PPBrowser, Page as PPPage, Target } from 'puppeteer'; +// `clickElements` and `clickElementsAndInterceptNavigationRequests` are internals of @crawlee/puppeteer +// and are deliberately not part of its public surface, so reach them through the source module. +import * as puppeteerClickElements from '../../../packages/puppeteer-crawler/src/internals/enqueue-links/click-elements.js'; import { runExampleComServer } from '../../shared/_helper.js'; function isPuppeteerBrowser(browser: PPBrowser | PWBrowser): browser is PPBrowser { @@ -25,6 +27,12 @@ function isPlaywrightBrowser(browser: PPBrowser | PWBrowser): browser is PWBrows return (browser as PWBrowser).browserType !== undefined; } +// Mirrors the module-local `isTargetRelevant` predicate in @crawlee/puppeteer's click-elements internals. +function isPuppeteerTargetRelevant(page: PPPage, target: Target): boolean { + // oxlint-disable-next-line typescript/no-deprecated -- the non-deprecated replacement (opener.page()) is async and cannot be awaited in this event callback + return target.type() === 'page' && page.target() === target.opener(); +} + async function createRequestQueueMock() { const enqueued: Source[] = []; const requestQueue = await RequestQueue.open({ id: 'xxx' }); @@ -504,8 +512,7 @@ testCases.forEach(({ caseName, launchBrowser, clickElements, utils }) => { }; (browser as PPBrowser).on('targetcreated', (target) => { counts.create++; - if ((clickElements as typeof puppeteerClickElements).isTargetRelevant(page, target)) - spawnedTarget = target; + if (isPuppeteerTargetRelevant(page, target)) spawnedTarget = target; }); browser.on('targetdestroyed', (target) => { counts.destroy++; diff --git a/test/core/error_snapshotter.test.ts b/test/core/error_snapshotter.test.ts index 78efe18cd4df..ed7286d4e841 100644 --- a/test/core/error_snapshotter.test.ts +++ b/test/core/error_snapshotter.test.ts @@ -1,29 +1,58 @@ -import type { KeyValueStore } from '@crawlee/core'; -import { ErrorSnapshotter } from '@crawlee/basic'; +import type { CrawlingContext } from '@crawlee/basic'; +import { ErrorTracker } from '@crawlee/basic'; import { describe, expect, test, vitest } from 'vitest'; -describe('ErrorSnapshotter', () => { - test('saveHTMLSnapshot returns the record key it stored the snapshot under', async () => { - const snapshotter = new ErrorSnapshotter(); +// `ErrorSnapshotter` is a module-internal collaborator of `ErrorTracker`; it is exercised +// through the only public entry point that reaches it - `new ErrorTracker({ saveErrorSnapshots: true })`. +describe('error snapshotter', () => { + // All grouping disabled, so the captured snapshot URLs land directly on `tracker.result`. + const newTracker = () => + new ErrorTracker({ + saveErrorSnapshots: true, + showStackTrace: false, + showErrorCode: false, + showErrorName: false, + showErrorMessage: false, + }); + + const contextWithStore = (keyValueStore: unknown) => + ({ + body: '', + getKeyValueStore: async () => keyValueStore, + }) as unknown as CrawlingContext; + + test('stores the HTML snapshot under the record key it resolves the public URL for', async () => { const setValue = vitest.fn(async () => {}); - const keyValueStore = { setValue } as unknown as KeyValueStore; + const getPublicUrl = vitest.fn(async (key: string) => `https://example.com/${key}`); + const tracker = newTracker(); - const key = await snapshotter.saveHTMLSnapshot('', keyValueStore, 'ERROR_SNAPSHOT_foo'); + await tracker.addAsync(new Error('some error'), contextWithStore({ setValue, getPublicUrl })); - expect(setValue).toHaveBeenCalledWith('ERROR_SNAPSHOT_foo', '', { contentType: 'text/html' }); - // The returned value must be the actual record key (no appended extension), - // as it is fed to `keyValueStore.getPublicUrl()`. - expect(key).toBe('ERROR_SNAPSHOT_foo'); + expect(setValue).toHaveBeenCalledTimes(1); + const [key, value, options] = setValue.mock.calls[0] as unknown as [string, string, unknown]; + expect(key).toMatch(/^ERROR_SNAPSHOT/); + expect(value).toBe(''); + expect(options).toEqual({ contentType: 'text/html' }); + // The public URL must be resolved for the actual record key (no appended extension). + expect(getPublicUrl).toHaveBeenCalledWith(key); + expect(tracker.result.firstErrorHtmlUrl).toBe(`https://example.com/${key}`); }); - test('saveHTMLSnapshot returns undefined when storing fails', async () => { - const snapshotter = new ErrorSnapshotter(); - const keyValueStore = { - setValue: async () => { - throw new Error('nope'); - }, - } as unknown as KeyValueStore; + test('records no snapshot URL when storing fails', async () => { + const getPublicUrl = vitest.fn(async (key: string) => `https://example.com/${key}`); + const tracker = newTracker(); + + await tracker.addAsync( + new Error('some error'), + contextWithStore({ + setValue: async () => { + throw new Error('nope'); + }, + getPublicUrl, + }), + ); - await expect(snapshotter.saveHTMLSnapshot('', keyValueStore, 'KEY')).resolves.toBeUndefined(); + expect(getPublicUrl).not.toHaveBeenCalled(); + expect(tracker.result.firstErrorHtmlUrl).toBeUndefined(); }); }); diff --git a/test/core/got_scraping_http_client.test.ts b/test/core/got_scraping_http_client.test.ts index 4532cbb19770..f5a88a6e925c 100644 --- a/test/core/got_scraping_http_client.test.ts +++ b/test/core/got_scraping_http_client.test.ts @@ -19,7 +19,7 @@ describe('GotScrapingHttpClient', () => { test('disables certificate verification when ignoreTlsErrors is set', async () => { const httpClient = new GotScrapingHttpClient(); - await httpClient.fetch(new Request('http://example.com'), { ignoreTlsErrors: true }); + await httpClient.sendRequest(new Request('http://example.com'), { ignoreTlsErrors: true }); expect(gotScraping).toHaveBeenCalledWith(expect.objectContaining({ https: { rejectUnauthorized: false } })); }); @@ -27,7 +27,7 @@ describe('GotScrapingHttpClient', () => { test('leaves certificate verification enabled without the flag', async () => { const httpClient = new GotScrapingHttpClient(); - await httpClient.fetch(new Request('http://example.com'), {}); + await httpClient.sendRequest(new Request('http://example.com')); expect(gotScraping).toHaveBeenCalledWith(expect.not.objectContaining({ https: expect.anything() })); }); diff --git a/test/core/impit_http_client.test.ts b/test/core/impit_http_client.test.ts index 58aa7648195f..e6cb1db47e80 100644 --- a/test/core/impit_http_client.test.ts +++ b/test/core/impit_http_client.test.ts @@ -14,20 +14,20 @@ describe('ImpitHttpClient', () => { vi.mocked(Impit).mockClear(); }); - test('reuses cached clients by default', () => { + test('reuses cached clients by default', async () => { const httpClient = new ImpitHttpClient(); - (httpClient as any).getClient({ proxyUrl: 'http://proxy.example' }); - (httpClient as any).getClient({ proxyUrl: 'http://proxy.example' }); + await httpClient.sendRequest(new Request('http://example.com')); + await httpClient.sendRequest(new Request('http://example.com')); expect(Impit).toHaveBeenCalledTimes(1); }); - test('creates a new client for each request when cacheClients is false', () => { + test('creates a new client for each request when cacheClients is false', async () => { const httpClient = new ImpitHttpClient({ cacheClients: false }); - (httpClient as any).getClient({ proxyUrl: 'http://proxy.example' }); - (httpClient as any).getClient({ proxyUrl: 'http://proxy.example' }); + await httpClient.sendRequest(new Request('http://example.com')); + await httpClient.sendRequest(new Request('http://example.com')); expect(Impit).toHaveBeenCalledTimes(2); }); @@ -35,7 +35,7 @@ describe('ImpitHttpClient', () => { test('forwards the per-request ignoreTlsErrors flag to the impit client', async () => { const httpClient = new ImpitHttpClient(); - await httpClient.fetch(new Request('http://example.com'), { ignoreTlsErrors: true }); + await httpClient.sendRequest(new Request('http://example.com'), { ignoreTlsErrors: true }); expect(Impit).toHaveBeenCalledWith(expect.objectContaining({ ignoreTlsErrors: true })); }); @@ -43,7 +43,7 @@ describe('ImpitHttpClient', () => { test('keeps constructor-level ignoreTlsErrors when the per-request flag is absent', async () => { const httpClient = new ImpitHttpClient({ ignoreTlsErrors: true }); - await httpClient.fetch(new Request('http://example.com'), {}); + await httpClient.sendRequest(new Request('http://example.com')); expect(Impit).toHaveBeenCalledWith(expect.objectContaining({ ignoreTlsErrors: true })); }); diff --git a/test/core/playwright_utils.test.ts b/test/core/playwright_utils.test.ts index 679e91dbc190..3964d3b1e7e2 100644 --- a/test/core/playwright_utils.test.ts +++ b/test/core/playwright_utils.test.ts @@ -439,8 +439,7 @@ describe('playwrightUtils', () => { }); describe('handleCloudflareChallenge() challenge detection', () => { - // not a named export, only reachable via the internal `playwrightUtils` object - const { handleCloudflareChallenge } = playwrightUtils.playwrightUtils; + const { handleCloudflareChallenge } = playwrightUtils; let browser: Browser; let page: Page; diff --git a/test/core/puppeteer_request_interception.test.ts b/test/core/puppeteer_request_interception.test.ts index e41cf0f41f30..87a830fd993d 100644 --- a/test/core/puppeteer_request_interception.test.ts +++ b/test/core/puppeteer_request_interception.test.ts @@ -1,13 +1,13 @@ import type { Server } from 'node:http'; import { sleep } from '@crawlee/utils'; -import { launchPuppeteer, utils } from 'crawlee'; +import { launchPuppeteer, puppeteerUtils } from 'crawlee'; // @ts-ignore This only throws when compiled against puppeteer 25+ (ESM only), we only import types, so its alllll gooooood import type { HTTPRequest } from 'puppeteer'; import { runExampleComServer } from '../shared/_helper.js'; -const { addInterceptRequestHandler, removeInterceptRequestHandler } = utils.puppeteer; +const { addInterceptRequestHandler, removeInterceptRequestHandler } = puppeteerUtils; let serverAddress = 'http://localhost:'; let port: number; @@ -21,7 +21,7 @@ beforeAll(async () => { afterAll(() => { server.close(); }); -describe('utils.puppeteer.addInterceptRequestHandler|removeInterceptRequestHandler()', () => { +describe('puppeteerUtils.addInterceptRequestHandler|removeInterceptRequestHandler()', () => { test('should allow multiple handlers', async () => { const browser = await launchPuppeteer({ launchOptions: { headless: true } }); @@ -209,7 +209,7 @@ describe('utils.puppeteer.addInterceptRequestHandler|removeInterceptRequestHandl }); }); -describe('utils.puppeteer.removeInterceptRequestHandler()', () => { +describe('puppeteerUtils.removeInterceptRequestHandler()', () => { test('works', async () => { const browser = await launchPuppeteer({ launchOptions: { headless: true } }); diff --git a/test/core/puppeteer_utils.test.ts b/test/core/puppeteer_utils.test.ts index f0b21c2acc8a..db26d77f38dd 100644 --- a/test/core/puppeteer_utils.test.ts +++ b/test/core/puppeteer_utils.test.ts @@ -3,9 +3,8 @@ import path from 'node:path'; import { MemoryStorageBackend, serviceLocator } from '@crawlee/core'; import { KeyValueStore, launchPuppeteer, puppeteerUtils, Request } from '@crawlee/puppeteer'; -import type { Dictionary } from '@crawlee/types'; // @ts-ignore This only throws when compiled against puppeteer 25+ (ESM only), we only import types, so its alllll gooooood -import type { Browser, Page, ResponseForRequest } from 'puppeteer'; +import type { Browser, Page } from 'puppeteer'; import { runExampleComServer } from '../shared/_helper.js'; import log from '@apify/log'; @@ -265,94 +264,6 @@ describe('puppeteerUtils', () => { ]), ); }); - - test('blockResources() supports default values', async () => { - const loadedUrls: string[] = []; - - const page = await browser.newPage(); - await puppeteerUtils.blockResources(page); - page.on('response', (response) => loadedUrls.push(response.url())); - await page.goto(`${serverAddress}/special/resources`, { waitUntil: 'load' }); - - expect(loadedUrls).toEqual(expect.arrayContaining([`${serverAddress}/script.js`])); - }); - - test('blockResources() supports nondefault values', async () => { - const loadedUrls: string[] = []; - - const page = await browser.newPage(); - await puppeteerUtils.blockResources(page, ['script']); - page.on('response', (response) => loadedUrls.push(response.url())); - await page.goto(`${serverAddress}/special/resources`, { waitUntil: 'load' }); - - expect(loadedUrls).toEqual( - expect.arrayContaining([`${serverAddress}/style.css`, `${serverAddress}/image.png`]), - ); - }); - }); - - test('supports cacheResponses()', async () => { - const browser = await launchPuppeteer(launchContext); - const cache: Dictionary> = {}; - - const getResourcesLoadedFromWiki = async () => { - let downloadedBytes = 0; - const page = await browser.newPage(); - page.setDefaultNavigationTimeout(0); - // Cache all javascript files, png files and svg files - await puppeteerUtils.cacheResponses(page, cache, ['.js', /.+\.png/i, /.+\.svg/i]); - page.on('response', async (response) => { - if (cache[response.url()]) return; - try { - const buffer = await response.buffer(); - downloadedBytes += buffer.byteLength; - } catch (e) { - // do nothing - } - }); - await page.goto(`${serverAddress}/cacheable`, { waitUntil: 'networkidle0', timeout: 60e3 }); - await page.close(); - return downloadedBytes; - }; - - try { - const bytesDownloadedOnFirstRun = await getResourcesLoadedFromWiki(); - const bytesDownloadedOnSecondRun = await getResourcesLoadedFromWiki(); - expect(bytesDownloadedOnSecondRun).toBeLessThan(bytesDownloadedOnFirstRun); - } finally { - await browser.close(); - } - }); - - test('cacheResponses() throws when rule with invalid type is provided', async () => { - const mockedPage = { - setRequestInterception: () => {}, - on: () => {}, - }; - - const testRuleType = async (value: string | RegExp) => { - try { - await puppeteerUtils.cacheResponses(mockedPage as any, {}, [value]); - } catch (error) { - // this is valid path for this test - return; - } - - expect(`Rule '${value}' should have thrown error`).toBe(''); - }; - - // @ts-expect-error - await testRuleType(0); - // @ts-expect-error - await testRuleType(1); - // @ts-expect-error - await testRuleType(null); - // @ts-expect-error - await testRuleType([]); - // @ts-expect-error - await testRuleType(['']); - // @ts-expect-error - await testRuleType(() => {}); }); test('compileScript() works', async () => { diff --git a/test/core/storages/dataset.test.ts b/test/core/storages/dataset.test.ts index f86c34870b4b..8307ed133afd 100644 --- a/test/core/storages/dataset.test.ts +++ b/test/core/storages/dataset.test.ts @@ -25,6 +25,7 @@ describe('dataset', () => { test('should work', async () => { const dataset = await Dataset.open(); + // @ts-expect-error Accessing private property const pushDataSpy = vitest.spyOn(dataset.backend, 'pushData'); const mockPushData = pushDataSpy.mockResolvedValueOnce(undefined); @@ -41,6 +42,7 @@ describe('dataset', () => { expect(mockPushData2).toHaveBeenCalledTimes(2); expect(mockPushData2).toHaveBeenCalledWith([{ foo: 'hotel;' }, { foo: 'restaurant' }]); + // @ts-expect-error Accessing private property const mockDrop = vitest.spyOn(dataset.backend, 'drop').mockResolvedValueOnce(undefined); await dataset.drop(); @@ -54,6 +56,7 @@ describe('dataset', () => { const dataset = await Dataset.open(); + // @ts-expect-error Accessing private property const mockPushData = vitest.spyOn(dataset.backend, 'pushData'); mockPushData.mockResolvedValueOnce(undefined); @@ -71,6 +74,7 @@ describe('dataset', () => { const dataset = await Dataset.open(); + // @ts-expect-error Accessing private property const mockPushData = vitest.spyOn(dataset.backend, 'pushData'); mockPushData.mockResolvedValueOnce(undefined); @@ -92,6 +96,7 @@ describe('dataset', () => { desc: false, }; + // @ts-expect-error Accessing private property const mockGetData = vitest.spyOn(dataset.backend, 'getData'); mockGetData.mockResolvedValueOnce(expected); @@ -104,6 +109,7 @@ describe('dataset', () => { expect(result).toEqual(expected); + // @ts-expect-error Accessing private property vitest.spyOn(dataset.backend, 'getData').mockImplementation(() => { throw new Error('Cannot create a string longer than 0x3fffffe7 characters'); }); @@ -124,6 +130,7 @@ describe('dataset', () => { itemCount: 14, }; + // @ts-expect-error Accessing private property const mockGetDataset = vitest.spyOn(dataset.backend, 'getMetadata'); mockGetDataset.mockResolvedValueOnce(expected); @@ -153,6 +160,7 @@ describe('dataset', () => { desc: false, }; + // @ts-expect-error Accessing private property const mockGetData = vitest.spyOn(dataset.backend, 'getData'); mockGetData.mockResolvedValueOnce(firstResolve); mockGetData.mockResolvedValueOnce(secondResolve); @@ -292,6 +300,7 @@ describe('dataset', () => { test('reduce() uses first value as memo if no memo is provided', async () => { const dataset = await Dataset.open(); + // @ts-expect-error Accessing private property const mockGetData = vitest.spyOn(dataset.backend, 'getData'); mockGetData.mockResolvedValueOnce({ items: [{ foo: 4 }, { foo: 5 }], diff --git a/test/core/storages/storage_purge.test.ts b/test/core/storages/storage_purge.test.ts index 991d4e5a2caf..927dc3cf27b0 100644 --- a/test/core/storages/storage_purge.test.ts +++ b/test/core/storages/storage_purge.test.ts @@ -121,9 +121,12 @@ describe('FileSystemStorageBackend.purge over a pre-existing storage directory', // input directory. Purging it would delete data we never wrote. test('leaves a storage directory it did not create alone', async () => { const foreignDirectory = temporaryDirectory(); + // The `` directories the backend writes under its `localDataDirectory` are part of the + // documented on-disk layout, so they are joined here rather than read back off the backend. + const keyValueStoresDirectory = resolve(foreignDirectory, 'key_value_stores'); const backend = new FileSystemStorageBackend({ localDataDirectory: foreignDirectory }); - await mkdir(resolve(backend.keyValueStoresDirectory, 'hand-placed'), { recursive: true }); - await writeFile(resolve(backend.keyValueStoresDirectory, 'hand-placed', 'INPUT.json'), '{"hand":"placed"}'); + await mkdir(resolve(keyValueStoresDirectory, 'hand-placed'), { recursive: true }); + await writeFile(resolve(keyValueStoresDirectory, 'hand-placed', 'INPUT.json'), '{"hand":"placed"}'); await backend.purge(); @@ -136,17 +139,18 @@ describe('FileSystemStorageBackend.purge over a pre-existing storage directory', // `{ id }` — not a run-scoped identifier, so it is not ours to empty. test('leaves an unnamed storage directory named after its own id alone', async () => { const ownDirectory = temporaryDirectory(); + const datasetsDirectory = resolve(ownDirectory, 'datasets'); const firstRun = new FileSystemStorageBackend({ localDataDirectory: ownDirectory }); const dataset = await firstRun.createDatasetBackend({ name: 'seed' }); await dataset.pushData([{ from: 'another-tool' }]); await firstRun.teardown(); // Turn the fixture into an unnamed storage living in an id-named directory. - const metadataPath = resolve(firstRun.datasetsDirectory, 'seed', '__metadata__.json'); + const metadataPath = resolve(datasetsDirectory, 'seed', '__metadata__.json'); const metadata = JSON.parse(await readFile(metadataPath, 'utf8')) as { id: string }; const { id } = metadata; await writeFile(metadataPath, JSON.stringify({ ...metadata, name: null })); - await rename(resolve(firstRun.datasetsDirectory, 'seed'), resolve(firstRun.datasetsDirectory, id)); + await rename(resolve(datasetsDirectory, 'seed'), resolve(datasetsDirectory, id)); const secondRun = new FileSystemStorageBackend({ localDataDirectory: ownDirectory }); await secondRun.purge(); diff --git a/test/core/storages/storage_transaction.test.ts b/test/core/storages/storage_transaction.test.ts index a9547a5eb698..76d927e2be9a 100644 --- a/test/core/storages/storage_transaction.test.ts +++ b/test/core/storages/storage_transaction.test.ts @@ -15,6 +15,7 @@ import { withStorageTransaction, } from '@crawlee/core'; import { BaseHttpClient } from '@crawlee/http-client'; +import type { DatasetBackend } from '@crawlee/types'; import { storage as timeoutStorage } from '@apify/timeout'; @@ -59,6 +60,7 @@ describe('StorageTransaction', () => { test('double commit does not double-flush', async () => { const dataset = await Dataset.open(); + // @ts-expect-error Accessing private property const pushDataSpy = vitest.spyOn(dataset.backend, 'pushData'); const transaction = createStorageTransaction(); @@ -73,6 +75,7 @@ describe('StorageTransaction', () => { test('a commit that throws lands in `failed` and later writes pass through', async () => { const dataset = await Dataset.open(); const store = await KeyValueStore.open(); + // @ts-expect-error Accessing private property vitest.spyOn(dataset.backend, 'pushData').mockRejectedValueOnce(new Error('backend exploded')); const transaction = createStorageTransaction(); @@ -126,6 +129,7 @@ describe('StorageTransaction', () => { test('receive the error of a failed commit, which still propagates', async () => { const dataset = await Dataset.open(); + // @ts-expect-error Accessing private property vitest.spyOn(dataset.backend, 'pushData').mockRejectedValueOnce(new Error('backend exploded')); const callback = vitest.fn(); @@ -143,6 +147,7 @@ describe('StorageTransaction', () => { test('an error thrown by a callback replaces the commit error', async () => { const dataset = await Dataset.open(); + // @ts-expect-error Accessing private property vitest.spyOn(dataset.backend, 'pushData').mockRejectedValueOnce(new Error('Data item is too large')); const transaction = createStorageTransaction(); @@ -319,6 +324,7 @@ describe('StorageTransaction', () => { // `failed` is the terminal state that neither `rollback()` nor an `open` check reaches - // dispose must release the journaled snapshots regardless of the outcome. + // @ts-expect-error Accessing private property vitest.spyOn(dataset.backend, 'pushData').mockRejectedValueOnce(new Error('boom')); await expect(transaction.commit()).rejects.toThrow('boom'); expect(transaction.state).toBe('failed'); @@ -391,8 +397,9 @@ describe('Dataset in a transaction', () => { // A backend honouring `skipEmpty` / `clean` / `unwind` returns fewer items than asked for while // the real items are far from exhausted - `total` stays honest, the page is just filtered. - const realBackend = dataset.backend; - dataset.backend = { + // @ts-expect-error Accessing private property + const realBackend: DatasetBackend<{ n: number }> = dataset.backend; + const stubBackend: DatasetBackend<{ n: number }> = { getMetadata: async () => realBackend.getMetadata(), drop: async () => realBackend.drop(), purge: async () => realBackend.purge(), @@ -403,6 +410,8 @@ describe('Dataset in a transaction', () => { return { ...page, items, count: items.length }; }, }; + // @ts-expect-error Accessing private property + dataset.backend = stubBackend; await withStorageTransaction(async (transaction) => { await dataset.pushData([{ n: 100 }, { n: 101 }]); @@ -426,6 +435,7 @@ describe('Dataset in a transaction', () => { test('values are captured at write time with full fidelity', async () => { const dataset = await Dataset.open(); + // @ts-expect-error Accessing private property const pushDataSpy = vitest.spyOn(dataset.backend, 'pushData'); const when = new Date('2023-01-01T00:00:00Z'); @@ -450,6 +460,7 @@ describe('Dataset in a transaction', () => { test('items from multiple pushData calls are committed in order, in a single backend call', async () => { const dataset = await Dataset.open(); + // @ts-expect-error Accessing private property const pushDataSpy = vitest.spyOn(dataset.backend, 'pushData'); await withStorageTransaction(async () => { From a1c8b84ac3c27d044f2c577b831cd8e013fac7be Mon Sep 17 00:00:00 2001 From: Jan Buchar Date: Wed, 19 Aug 2026 11:12:23 +0200 Subject: [PATCH 02/16] refactor!: Collapse the context pipeline seam into `contextPipelineBuilder` MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `BasicCrawler.buildContextPipeline` only ever returned an empty pipeline, so it is gone and the eight subclasses that extended it now keep their builders `#private` — `HttpCrawler` and `BrowserCrawler` stay `protected`, being the only levels that genuinely compose stages. `FileDownload` also stops silently discarding a user-supplied `contextPipelineBuilder`. --- docs/public-api/crawlee-basic.api.md | 1 - docs/public-api/crawlee-puppeteer.api.md | 4 --- docs/public-api/crawlee-stagehand.api.md | 4 --- docs/upgrading/upgrading_v4.md | 4 ++- .../src/internals/basic-crawler.ts | 10 +----- .../src/internals/browser-crawler.ts | 3 +- .../src/internals/cheerio-crawler.ts | 5 +++ .../src/internals/file-download.ts | 25 +++++++------ .../src/internals/http-crawler.ts | 3 +- .../internals/adaptive-playwright-crawler.ts | 8 ++--- .../src/internals/playwright-crawler.ts | 4 +-- .../src/internals/puppeteer-crawler.ts | 4 +-- .../src/internals/stagehand-crawler.ts | 4 +-- test/core/crawlers/file_download.test.ts | 35 ++++++++++++++++++- 14 files changed, 69 insertions(+), 45 deletions(-) diff --git a/docs/public-api/crawlee-basic.api.md b/docs/public-api/crawlee-basic.api.md index 2dea2533e4cb..57c7c3af7597 100644 --- a/docs/public-api/crawlee-basic.api.md +++ b/docs/public-api/crawlee-basic.api.md @@ -60,7 +60,6 @@ export class BasicCrawler>, options?: CrawlerAddRequestsOptions): Promise; // (undocumented) protected blockedStatusCodes: Set; - protected buildContextPipeline(): ContextPipeline; get concurrencySystem(): IConcurrencySystem | undefined; protected createDefaultConcurrencySystem(options: ConcurrencySystemOptions): ConcurrencySystem; destroy(): Promise; diff --git a/docs/public-api/crawlee-puppeteer.api.md b/docs/public-api/crawlee-puppeteer.api.md index 3f368c9c1dc2..6fa4ac05390e 100644 --- a/docs/public-api/crawlee-puppeteer.api.md +++ b/docs/public-api/crawlee-puppeteer.api.md @@ -17,8 +17,6 @@ import type { BrowserPoolOptions } from '@crawlee/browser-pool'; import type { CheerioAPI } from 'cheerio'; import type { ClickOptions } from 'puppeteer'; import { Configuration } from '@crawlee/browser'; -import type { ContextPipeline } from '@crawlee/browser'; -import type { CrawlingContext } from '@crawlee/browser'; import { Dictionary } from '@crawlee/types'; import type { GetUserDataFromRequest } from '@crawlee/browser'; import type { HTTPRequest } from 'puppeteer'; @@ -165,8 +163,6 @@ interface PuppeteerContextUtils { export class PuppeteerCrawler, ExtendedContext extends PuppeteerCrawlingContext = PuppeteerCrawlingContext & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends object = {}> extends BrowserCrawler { constructor(options?: PuppeteerCrawlerOptions); // (undocumented) - protected buildContextPipeline(): ContextPipeline; - // (undocumented) protected navigationHandler(crawlingContext: PuppeteerCrawlingContext, gotoOptions: PuppeteerDirectNavigationOptions): Promise; } diff --git a/docs/public-api/crawlee-stagehand.api.md b/docs/public-api/crawlee-stagehand.api.md index 618da744e204..c47a8cf870b5 100644 --- a/docs/public-api/crawlee-stagehand.api.md +++ b/docs/public-api/crawlee-stagehand.api.md @@ -22,8 +22,6 @@ import type { BrowserPoolHooks } from '@crawlee/browser-pool'; import type { BrowserPoolOptions } from '@crawlee/browser-pool'; import type { BrowserType } from 'playwright'; import { Configuration } from '@crawlee/browser'; -import type { ContextPipeline } from '@crawlee/browser'; -import type { CrawlingContext } from '@crawlee/browser'; import type { Dictionary } from '@crawlee/types'; import { ExtractOptions } from '@browserbasehq/stagehand'; import type { GetUserDataFromRequest } from '@crawlee/browser'; @@ -97,8 +95,6 @@ export interface StagehandBrowserPoolOptions extends Omit, ExtendedContext extends StagehandCrawlingContext = StagehandCrawlingContext & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends object = {}> extends BrowserCrawler { constructor(options?: StagehandCrawlerOptions); - // (undocumented) - protected buildContextPipeline(): ContextPipeline; protected navigationHandler(crawlingContext: StagehandCrawlingContext, gotoOptions: StagehandGotoOptions): Promise; } diff --git a/docs/upgrading/upgrading_v4.md b/docs/upgrading/upgrading_v4.md index 560adbf2509f..d43d752fe6ca 100644 --- a/docs/upgrading/upgrading_v4.md +++ b/docs/upgrading/upgrading_v4.md @@ -977,7 +977,9 @@ These are still present at runtime, but they are excluded from the documented su `HttpCrawler.isRequestBlocked` and `FileDownload.buildContextPipeline` are now `private`. Block detection is configured through `retryOnBlocked` and `blockedStatusCodes`; to add your own checks, throw a `SessionError` from a `postNavigationHook`. To extend the `FileDownload` context pipeline, pass a `contextPipelineBuilder` to the constructor rather than subclassing. -`HttpCrawler.buildContextPipeline`, `HttpCrawler.getNavigationTimeoutMillis`, `HttpCrawler.createDefaultConcurrencySystem` and `JSDOMCrawler.buildContextPipeline` stay `protected` and overridable, but are now marked `@internal` — they are implementation seams shared with `@crawlee/cheerio`, `@crawlee/jsdom` and `@crawlee/linkedom`, and their signatures may change in a minor release. +`HttpCrawler.buildContextPipeline` and `BrowserCrawler.buildContextPipeline` stay `protected` and overridable, and are supported extension points: they are the two levels that genuinely compose context stages, so override one of them to add your own. `BasicCrawler.buildContextPipeline` is gone — it only ever returned an empty pipeline — and the builders on `CheerioCrawler`, `JSDOMCrawler`, `LinkeDOMCrawler`, `PlaywrightCrawler`, `AdaptivePlaywrightCrawler`, `PuppeteerCrawler` and `StagehandCrawler` are now `#private`. If you were overriding one of those, pass a `contextPipelineBuilder` to the constructor instead. + +`HttpCrawler.getNavigationTimeoutMillis` and `HttpCrawler.createDefaultConcurrencySystem` stay `protected` and overridable, but are marked `@internal` — they are implementation seams shared with `@crawlee/cheerio`, `@crawlee/jsdom` and `@crawlee/linkedom`, and their signatures may change in a minor release. ### The protected `getMessageFromError()` returns `string` diff --git a/packages/basic-crawler/src/internals/basic-crawler.ts b/packages/basic-crawler/src/internals/basic-crawler.ts index f2cfe4fd4844..a372f4bf9eed 100644 --- a/packages/basic-crawler/src/internals/basic-crawler.ts +++ b/packages/basic-crawler/src/internals/basic-crawler.ts @@ -1442,14 +1442,6 @@ export class BasicCrawler< return {}; } - /** - * Builds the subclass-specific context pipeline that transforms a `CrawlingContext` into the crawler's target context type. - * Subclasses should override this to add their own pipeline stages. - */ - protected buildContextPipeline(): ContextPipeline { - return ContextPipeline.create(); - } - private createBaseContext(context: PendingCrawlingContext) { const deferredCleanup: (() => Promise)[] = []; @@ -1540,7 +1532,7 @@ export class BasicCrawler< private buildFinalContextPipeline(): ContextPipeline { const subclassPipeline = (this.#contextPipelineOptions.contextPipelineBuilder?.() ?? - this.buildContextPipeline()) as ContextPipeline; + ContextPipeline.create()) as ContextPipeline; // `extendContext` runs *before* the subclass navigation pipeline (which includes the // pre/post-navigation hooks). This makes the extension visible to those hooks and to the diff --git a/packages/browser-crawler/src/internals/browser-crawler.ts b/packages/browser-crawler/src/internals/browser-crawler.ts index 884c9a9b0065..32fb6451e704 100644 --- a/packages/browser-crawler/src/internals/browser-crawler.ts +++ b/packages/browser-crawler/src/internals/browser-crawler.ts @@ -501,8 +501,7 @@ export abstract class BrowserCrawler< return this.#navigationTimeoutMillis; } - /** @internal */ - protected override buildContextPipeline(): ContextPipeline< + protected buildContextPipeline(): ContextPipeline< CrawlingContext, BrowserCrawlingContext > { diff --git a/packages/cheerio-crawler/src/internals/cheerio-crawler.ts b/packages/cheerio-crawler/src/internals/cheerio-crawler.ts index 0b642dde2b39..16d81fcac6f8 100644 --- a/packages/cheerio-crawler/src/internals/cheerio-crawler.ts +++ b/packages/cheerio-crawler/src/internals/cheerio-crawler.ts @@ -38,6 +38,11 @@ export interface CheerioCrawlerOptions< StatisticStateExtension > {} +export type CheerioHook< + UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler + JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler +> = InternalHttpHook>; + export interface CheerioCrawlingContext< UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler diff --git a/packages/http-crawler/src/internals/file-download.ts b/packages/http-crawler/src/internals/file-download.ts index 389ca590ec16..bf68b2d0caec 100644 --- a/packages/http-crawler/src/internals/file-download.ts +++ b/packages/http-crawler/src/internals/file-download.ts @@ -1,7 +1,5 @@ -import { Transform } from 'node:stream'; - -import type { BasicCrawlerOptions, ContextPipeline, CrawlingContext, LoadedRequest } from '@crawlee/basic'; -import { BasicCrawler } from '@crawlee/basic'; +import type { BasicCrawlerOptions, CrawlingContext, LoadedRequest } from '@crawlee/basic'; +import { BasicCrawler, ContextPipeline } from '@crawlee/basic'; import type { Request } from '@crawlee/core'; import { ResponseWithUrl } from '@crawlee/http-client'; import type { Dictionary } from '@crawlee/types'; @@ -52,14 +50,19 @@ export type FileDownloadRequestHandler< * * The crawler finishes when there are no more {@apilink Request} objects to crawl. * - * We can use the `preNavigationHooks` to adjust the crawling context before the request is made: + * `FileDownload` has no `preNavigationHooks` / `postNavigationHooks` options. To adjust the crawling context before + * the request is made, pass your own {@apilink BasicCrawlerOptions.contextPipelineBuilder|`contextPipelineBuilder`}: * - * ``` - * preNavigationHooks: [ - * (crawlingContext) => { + * ```ts + * const crawler = new FileDownload({ + * contextPipelineBuilder: () => + * ContextPipeline.create().compose({ + * action: async (context) => ({ ...context, myField: 123 }), + * }), + * requestHandler({ myField }) { * // ... * }, - * ] + * }); * ``` * * New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's {@apilink ConcurrencySystem}. Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the `FileCrawler` constructor, or, for finer control, by injecting a pre-configured {@apilink ConcurrencySystem|`concurrencySystem`}. @@ -85,12 +88,12 @@ export class FileDownload extends BasicCrawler { constructor(options: BasicCrawlerOptions = {}) { super({ ...options, - contextPipelineBuilder: () => this.#buildContextPipeline(), + contextPipelineBuilder: options.contextPipelineBuilder ?? (() => this.#buildContextPipeline()), }); } #buildContextPipeline(): ContextPipeline { - return super.buildContextPipeline().compose({ + return ContextPipeline.create().compose({ action: async (context) => this.initiateDownload(context), cleanup: async (context) => { if (!context.response.bodyUsed) { diff --git a/packages/http-crawler/src/internals/http-crawler.ts b/packages/http-crawler/src/internals/http-crawler.ts index 699c8a2c96c5..7cb869601a57 100644 --- a/packages/http-crawler/src/internals/http-crawler.ts +++ b/packages/http-crawler/src/internals/http-crawler.ts @@ -449,8 +449,7 @@ export class HttpCrawler< }); } - /** @internal */ - protected override buildContextPipeline(): ContextPipeline { + protected buildContextPipeline(): ContextPipeline { // When navigation is skipped, `prepareHttpRequest` has already installed throwing getters for // the response-derived members, so the guarded action is bypassed and the context left untouched. const skipGuard = ( diff --git a/packages/playwright-crawler/src/internals/adaptive-playwright-crawler.ts b/packages/playwright-crawler/src/internals/adaptive-playwright-crawler.ts index 648bdda4e3fb..ad3b2e162f1e 100644 --- a/packages/playwright-crawler/src/internals/adaptive-playwright-crawler.ts +++ b/packages/playwright-crawler/src/internals/adaptive-playwright-crawler.ts @@ -11,7 +11,6 @@ import { isDeepStrictEqual } from 'node:util'; import type { BasicCrawlerOptions, - ContextPipeline, CrawlingContext, EnqueueLinksOptions, GetUserDataFromRequest, @@ -21,6 +20,7 @@ import type { } from '@crawlee/basic'; import { BasicCrawler, + ContextPipeline, RequestHandlerError, resolveBaseUrlForEnqueueLinksFiltering, Router, @@ -426,7 +426,7 @@ export class AdaptivePlaywrightCrawler< stateExtension: adaptivePlaywrightCrawlerStatisticState as StatisticStateExtensionOptions, }), - contextPipelineBuilder: contextPipelineBuilder ?? (() => this.buildContextPipeline()), + contextPipelineBuilder: contextPipelineBuilder ?? (() => this.#buildContextPipeline()), // The base crawler must not wrap requests in a transaction of its own - this crawler opens // one per request handler attempt in `crawlOne` instead, forwarding the write policy of the // user-facing option (validated above) to those. @@ -512,11 +512,11 @@ export class AdaptivePlaywrightCrawler< return await super.init(); } - protected override buildContextPipeline(): ContextPipeline { + #buildContextPipeline(): ContextPipeline { const errorMessage = (prop: string) => `The \`${prop}\` property is not available on the outer context pipeline of AdaptivePlaywrightCrawler - it is provided by the inner (static/browser) pipelines`; - return super.buildContextPipeline().compose({ + return ContextPipeline.create().compose({ action: async ({ request }) => ({ get request(): LoadedRequest> { return request as LoadedRequest>; diff --git a/packages/playwright-crawler/src/internals/playwright-crawler.ts b/packages/playwright-crawler/src/internals/playwright-crawler.ts index f29d18b13125..69a532934fd1 100644 --- a/packages/playwright-crawler/src/internals/playwright-crawler.ts +++ b/packages/playwright-crawler/src/internals/playwright-crawler.ts @@ -254,11 +254,11 @@ export class PlaywrightCrawler< remoteBrowser ? remotePlaywrightBrowserPool({ ...remoteBrowser, launchContext, headless, configuration }) : playwrightBrowserPool({ launchContext, headless, configuration }), - contextPipelineBuilder: contextPipelineBuilder ?? (() => this.buildContextPipeline()), + contextPipelineBuilder: contextPipelineBuilder ?? (() => this.#buildContextPipeline()), }); } - protected override buildContextPipeline(): ContextPipeline { + #buildContextPipeline(): ContextPipeline { return super.buildContextPipeline().compose({ action: this.enhanceContext.bind(this) }); } diff --git a/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts b/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts index 23228af0249d..846c3ab063ac 100644 --- a/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts +++ b/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts @@ -257,11 +257,11 @@ export class PuppeteerCrawler< remoteBrowser ? remotePuppeteerBrowserPool({ ...remoteBrowser, launchContext, headless, configuration }) : puppeteerBrowserPool({ launchContext, headless, configuration }), - contextPipelineBuilder: contextPipelineBuilder ?? (() => this.buildContextPipeline()), + contextPipelineBuilder: contextPipelineBuilder ?? (() => this.#buildContextPipeline()), }); } - protected override buildContextPipeline(): ContextPipeline { + #buildContextPipeline(): ContextPipeline { return super.buildContextPipeline().compose({ action: this.enhanceContext.bind(this) }); } diff --git a/packages/stagehand-crawler/src/internals/stagehand-crawler.ts b/packages/stagehand-crawler/src/internals/stagehand-crawler.ts index 1f438be4b953..ce7bc1406f16 100644 --- a/packages/stagehand-crawler/src/internals/stagehand-crawler.ts +++ b/packages/stagehand-crawler/src/internals/stagehand-crawler.ts @@ -471,11 +471,11 @@ export class StagehandCrawler< headless, configuration, })) as unknown as OwnedBrowserPool, - contextPipelineBuilder: contextPipelineBuilder ?? (() => this.buildContextPipeline()), + contextPipelineBuilder: contextPipelineBuilder ?? (() => this.#buildContextPipeline()), }); } - protected override buildContextPipeline(): ContextPipeline { + #buildContextPipeline(): ContextPipeline { return super.buildContextPipeline().compose({ action: this.setUpStagehand.bind(this) }); } diff --git a/test/core/crawlers/file_download.test.ts b/test/core/crawlers/file_download.test.ts index 06751a99606d..186ac5f6c4ca 100644 --- a/test/core/crawlers/file_download.test.ts +++ b/test/core/crawlers/file_download.test.ts @@ -5,7 +5,8 @@ import { pipeline } from 'node:stream/promises'; import { ReadableStream } from 'node:stream/web'; import { setTimeout } from 'node:timers/promises'; -import { FileDownload } from '@crawlee/http'; +import type { CrawlingContext, LoadedRequest, Request } from '@crawlee/http'; +import { ContextPipeline, FileDownload } from '@crawlee/http'; import { FetchHttpClient } from '@crawlee/http-client'; import express from 'express'; import { startExpressAppPromise } from '../../shared/_helper.js'; @@ -200,3 +201,35 @@ test('crawler waits for the stream to be consumed', async () => { expect(bufferedData.length).toBe(5 * 1024); expect(bufferedData).toEqual(await ReadableStreamGenerator.getUint8Array(5 * 1024, 789)); }); + +test('honours a user-supplied contextPipelineBuilder', async () => { + let builderCalls = 0; + const bodies: string[] = []; + + const crawler = new FileDownload({ + maxRequestRetries: 0, + contextPipelineBuilder: () => { + builderCalls++; + + return ContextPipeline.create().compose({ + action: async (context) => ({ + request: context.request as LoadedRequest, + response: new Response('stubbed'), + contentType: { type: 'text/plain', encoding: 'utf8' as BufferEncoding }, + }), + }); + }, + requestHandler: async ({ response }) => { + bodies.push(await response.text()); + }, + }); + + const fileUrl = new URL('/file?size=1024&seed=123', url).toString(); + + const stats = await crawler.run([fileUrl]); + + expect(stats.requestsFailed).toBe(0); + expect(builderCalls).toBe(1); + // The supplied pipeline replaces the built-in download, so the handler sees the stub, not the file. + expect(bodies).toEqual(['stubbed']); +}); From 1a3730756d99c1f547d3307c3ed5515927ed77d8 Mon Sep 17 00:00:00 2001 From: Jan Buchar Date: Wed, 19 Aug 2026 22:26:47 +0200 Subject: [PATCH 03/16] refactor!: Make BasicCrawler.requestManager read-only --- docs/public-api/crawlee-basic.api.md | 2 +- docs/upgrading/upgrading_v4.md | 25 ++++++++++++++++++- .../src/internals/basic-crawler.ts | 18 +++++++------ test/core/crawlers/basic_crawler.test.ts | 10 +++++--- 4 files changed, 42 insertions(+), 13 deletions(-) diff --git a/docs/public-api/crawlee-basic.api.md b/docs/public-api/crawlee-basic.api.md index 57c7c3af7597..369db00ffa7a 100644 --- a/docs/public-api/crawlee-basic.api.md +++ b/docs/public-api/crawlee-basic.api.md @@ -89,7 +89,7 @@ export class BasicCrawler; - protected requestManager?: IRequestManager; + protected get requestManager(): IRequestManager | undefined; resume(): void; // (undocumented) protected readonly retryOnBlocked: boolean; diff --git a/docs/upgrading/upgrading_v4.md b/docs/upgrading/upgrading_v4.md index d43d752fe6ca..229133f85b0c 100644 --- a/docs/upgrading/upgrading_v4.md +++ b/docs/upgrading/upgrading_v4.md @@ -1739,7 +1739,30 @@ It is also what enforces robots.txt `Crawl-delay` directives — with `respectRo #### `BasicCrawler.requestList` and `BasicCrawler.requestQueue` fields removed -The public `requestList` and `requestQueue` instance fields are gone. The crawler exposes a single `protected requestManager?: IRequestManager` instead. Access the active manager via the new async `getRequestManager()` method. +The public `requestList` and `requestQueue` instance fields are gone. The crawler exposes a single read-only `protected requestManager` getter instead. Access the active manager via the new async `getRequestManager()` method. + +#### `BasicCrawler.requestManager` is read-only + +`requestManager` is now a getter over a native `#requestManager` field, so a subclass can read it but can no longer assign to it. The crawler owns the manager's lifecycle — it resolves the `requestManager` / `requestList` / `requestQueue` options, opens a default queue when none was given, and wraps the result in a `ThrottlingRequestManager` when `sameDomainDelaySecs` is set — and assigning over it from a subclass skipped those steps. + +Inject your own manager through the constructor option instead, and read the resolved one with `getRequestManager()`: + +**Before:** +```typescript +class MyCrawler extends BasicCrawler { + protected override async init() { + await super.init(); + this.requestManager = await MyRequestQueue.open(); + } +} +``` + +**After:** +```typescript +const crawler = new MyCrawler({ requestManager: await MyRequestQueue.open() }); +``` + +Being a native `#` field, it is also no longer visible to `Object.keys()`, object spread or `JSON.stringify()`. #### `getRequestQueue()` deprecated in favor of `getRequestManager()` diff --git a/packages/basic-crawler/src/internals/basic-crawler.ts b/packages/basic-crawler/src/internals/basic-crawler.ts index a372f4bf9eed..7939f4dd9be3 100644 --- a/packages/basic-crawler/src/internals/basic-crawler.ts +++ b/packages/basic-crawler/src/internals/basic-crawler.ts @@ -727,12 +727,16 @@ export class BasicCrawler< return this.#statisticsDep.value; } + #requestManager?: IRequestManager; + /** * The main request-handling component of the crawler. It manages the requests that the crawler processes, * combining any provided request loader and/or queue. It's initialized during the crawler startup or lazily * via {@apilink BasicCrawler.getRequestManager|`getRequestManager()`}. */ - protected requestManager?: IRequestManager; + protected get requestManager(): IRequestManager | undefined { + return this.#requestManager; + } /** Backs the {@apilink BasicCrawler.sessionPool|`sessionPool`} getter. */ #sessionPoolDep: OwnedOrInjected; @@ -1105,13 +1109,13 @@ export class BasicCrawler< if (requestList !== undefined) { // The list is read first, while new requests still have somewhere writable to go. - this.requestManager = new RequestManagerTandem( + this.#requestManager = new RequestManagerTandem( requestList, writableManager ?? (() => this.openOwnedRequestQueue()), ); } else if (writableManager !== undefined) { // A RequestQueue is itself a request manager. - this.requestManager = writableManager; + this.#requestManager = writableManager; } this.httpClient = httpClient ?? new LazyDefaultHttpClient({ logger: this.log }); @@ -1909,8 +1913,8 @@ export class BasicCrawler< * if none has been configured or opened yet. */ async getRequestManager(): Promise { - if (!this.requestManager) { - this.requestManager = await this.openOwnedRequestQueue(); + if (!this.#requestManager) { + this.#requestManager = await this.openOwnedRequestQueue(); } // Apply the processing-time hint here (an async lifecycle point) rather than in the constructor, @@ -1918,10 +1922,10 @@ export class BasicCrawler< // but guard so we do not re-issue it on every call. if (!this.#requestManagerTimeoutsApplied) { this.#requestManagerTimeoutsApplied = true; - await this.applyRequestManagerTimeouts(this.requestManager); + await this.applyRequestManagerTimeouts(this.#requestManager); } - return this.requestManager; + return this.#requestManager; } /** diff --git a/test/core/crawlers/basic_crawler.test.ts b/test/core/crawlers/basic_crawler.test.ts index 9113a1ccee84..e69e37a67780 100644 --- a/test/core/crawlers/basic_crawler.test.ts +++ b/test/core/crawlers/basic_crawler.test.ts @@ -613,8 +613,7 @@ describe('BasicCrawler', () => { let drainedRequests: any[]; let options: EnqueueLinksOptions; let requestQueue: RequestQueue; - - const crawler = new BasicCrawler({ maxCrawlDepth: 3 }); + let crawler: BasicCrawler; // Mimics what `context.addRequests()` would have tagged the URLs with, based on the current // request's `crawlDepth`. @@ -637,9 +636,12 @@ describe('BasicCrawler', () => { }; requestQueue = { addRequestsBatched: addRequestsBatchedMock as RequestQueue['addRequestsBatched'], + // Only `addRequestsBatched` is exercised here; the other two exist because the + // `requestManager` option is validated structurally. + fetchNextRequest: (async () => null) as unknown as RequestQueue['fetchNextRequest'], + addRequest: (async () => ({})) as unknown as RequestQueue['addRequest'], } as RequestQueue; - // eslint-disable-next-line dot-notation -- private field on the crawler, injected for the mock - crawler['requestManager'] = requestQueue; + crawler = new BasicCrawler({ maxCrawlDepth: 3, requestManager: requestQueue }); }); it('should generate requests with maxCrawlDepth', async () => { From 4ae206ab7d9dc1cdfb9fea1f0d16dea78adf7973 Mon Sep 17 00:00:00 2001 From: Jan Buchar Date: Thu, 20 Aug 2026 21:29:39 +0200 Subject: [PATCH 04/16] Add a guide about the public API export and our BC guarantees --- docs/guides/public_api.mdx | 58 ++++++++++++++++++++++++++++++++++++ docs/public-api/README.md | 61 +++++++++++++++++++++++++++++++++----- 2 files changed, 111 insertions(+), 8 deletions(-) create mode 100644 docs/guides/public_api.mdx diff --git a/docs/guides/public_api.mdx b/docs/guides/public_api.mdx new file mode 100644 index 000000000000..ecb7e544c9b7 --- /dev/null +++ b/docs/guides/public_api.mdx @@ -0,0 +1,58 @@ +--- +id: public-api +title: Public API +description: What Crawlee promises not to break, and what it doesn't +--- + +Crawlee ships its type definitions, and those definitions contain more than its supported API. +Some of what you can import is a deliberate promise; some of it is machinery that happens to be +reachable. This page explains how to tell, so you can decide what to depend on. + +## Supported by default + +Anything Crawlee exports is supported unless it says otherwise. If you can import it and its +documentation does not mark it as internal, it is covered by backwards compatibility: it will +not change shape or disappear outside a major release, and if it ever does, the change is +recorded in the upgrading guide. + +## Marked internal + +Some members carry an `@internal` tag in their documentation comment. Your editor shows it when +you hover the symbol, and it means exactly one thing: **we do not promise anything about it.** +It can change signature, behaviour or disappear entirely in any release, including a patch, and +it will not appear in the upgrading guide when it does. + +It is still exported, still typed, and auto-completion still works. That is on purpose. We +would rather leave you a way to unblock yourself — knowingly — than take it away and have you +patch the package or give up. So the tag is a statement about support, not about access: + +```ts +// Fine. Supported, and it will keep working. +import { CheerioCrawler, Dataset } from 'crawlee'; + +// Allowed, but you are on your own. Pin your Crawlee version +// and expect to revisit this on every upgrade. +import { someInternalHelper } from 'crawlee'; +``` + +If you find yourself reaching for an internal member to get something done, that is worth +telling us about — [open an issue](https://github.com/apify/crawlee/issues). Those reports are +what we use to decide which extension points deserve a real, supported API. + +## Things that are not types + +Not every promise is expressible in TypeScript, and a few things are contracts even though +nothing checks them: + +- **Persisted state.** The layout of what Crawlee writes into a key-value store or request + queue is an implementation detail. Read it for debugging, do not build on it. +- **Log message text.** Messages change freely; never match on them. +- **Subclass hook ordering.** When you override a documented extension point, call `super` + where the base class expects it. Skipping it usually compiles and then misbehaves at runtime. + +## Where to check + +For anything beyond the obvious, the per-package surface maps under +[`docs/public-api/`](https://github.com/apify/crawlee/tree/master/docs/public-api) in the +repository are the authoritative inventory of what we promise. If a symbol is in there, it is +supported; if not, it is not. diff --git a/docs/public-api/README.md b/docs/public-api/README.md index 2a92effa0579..ea6e99ec9195 100644 --- a/docs/public-api/README.md +++ b/docs/public-api/README.md @@ -8,6 +8,30 @@ promise backwards compatibility**. They are produced by [API Extractor](https://api-extractor.com/) from the built `dist/index.d.ts` of each package. +## What a tag means + +These reports are an **inventory of backwards-compatibility promises**. Membership is the +promise; the tags are how a member joins or leaves it. + +- **Untagged — promised.** In the report, and covered by backwards compatibility. The codebase + does not use explicit `@public` tags, so untagged is the default and anything you add is + promised unless you say otherwise. +- **`@internal` (and the legacy `@ignore`) — not promised.** Trimmed from the report. It is + still exported, still present in the published `.d.ts`, and still has working + auto-completion. You may use it; it may change or disappear in any release, including a + patch. + +**We deliberately do not strip these members from the `.d.ts`.** No `stripInternal`, no +parallel "internal build". Two reasons: crawlee's own packages consume plenty of each other's +internals, and an escape hatch that lets someone unblock themselves — knowingly, at their own +risk — is worth more than one we have taken away. So the tag is a statement about *support*, +not about *access*. + +The practical consequence when reviewing: tagging something `@internal` does not make it +harder to reach, it only stops us owing anyone stability on it. Reach for `#private` or TS +`private` when a member genuinely should be unreachable, and for `@internal` when it should be +reachable but unsupported. Both are legitimate; they answer different questions. + ## Workflow - After changing any package's public surface, regenerate the reports and commit them: @@ -35,11 +59,12 @@ They are produced by [API Extractor](https://api-extractor.com/) from the built ## Notes -- The reports are generated as API Extractor's **`public`** variant, so symbols tagged - `@internal` (`@alpha`/`@beta` too) are excluded — only `@public` surface is tracked. - The legacy `@ignore` tag counts as `@internal` here; the generator rewrites it before - extraction, so an `@ignore`-d symbol is excluded too and cannot be referenced from a - `@public` signature. +- Mechanically, the reports are API Extractor's **`public`** variant, so `@internal` + (`@alpha`/`@beta` too) is excluded. `@ignore` is a TypeDoc-era tag that API Extractor does + not act on, so the generator rewrites it to `@internal` before extraction — see `IGNORE_TAG` + in `scripts/api-extractor/run.ts`. It rewrites nothing else, which is worth knowing: + **`@private` is inert here.** A member whose only tag is `@private` stays in the report and + stays promised, so it is not a way to de-promise anything — use `@internal`. The generator stages the variant as `.public.api.md` under `temp/` and promotes it onto the committed `.api.md`, so the tracked filenames stay stable. - API Extractor builds the import list before it trims the non-`@public` declarations and @@ -73,11 +98,31 @@ They are produced by [API Extractor](https://api-extractor.com/) from the built - `@crawlee/cli` and `@crawlee/templates` are deliberately excluded — they are tooling (a CLI binary and project scaffolding), not an importable API where we promise BC. The exclude list lives in `scripts/api-extractor/run.ts`. +- `crawlee.api.md` looks almost empty — eleven `export *` lines and no declarations — and that + is correct rather than a gap. The meta-package re-exports; it declares nothing of its own. + Everything it hands you belongs to a constituent package and is already inventoried there, so + a break in any of it shows up in that package's report first. Only two kinds of change are + meta-package-only, and this report catches both: dropping an `export *` line, and changing + something the meta-package declares itself — the `utils` bag that used to live here was in + the report, and its removal showed up as a diff. Listing the ~380 re-exported names instead + would restate promises where they are not made and churn on every constituent change. + + The case that looks like a hole is not one. If two constituents ever export *different* + symbols under the same name, TypeScript raises `TS2308` ("has already exported a member + named …") in `packages/crawlee/src/index.ts`, so the build fails — the name does not quietly + drop out of the barrel. Several hundred names are re-exported by more than one constituent + today; every one of them is a single symbol reached by several paths, which is fine and + raises nothing. - The generator lives in `scripts/api-extractor/`. It temporarily strips the build's injected `// @ts-ignore` comment lines from the `.d.ts` files (restoring them afterwards) because API Extractor's AST walker trips over some of them; a small number of packages additionally need a sanitized-mirror fallback. See the comments in `scripts/api-extractor/run.ts` for details. -- These reports now cover only the `@public` surface. Further shrinking them — genuinely - hiding class internals (untagged `protected`/`_`-prefixed members) rather than merely - tagging them — is the goal tracked in issue #3109. +- **The reports are an inventory of what we promise, not a measure of how much code is + reachable.** A symbol is in a report because we have committed to not breaking it; a symbol + is absent because we have not. Absent does *not* mean inaccessible, and making it + inaccessible is explicitly not the goal — see "What a tag means" above. Shrinking a report + is therefore only ever a *consequence* of deciding that something was never a promise, never + a target in its own right. A change that removes entries without changing any decision has + achieved nothing; a change that adds entries because we decided to support something is a + success. Issue #3109 tracks that decision-making, not a line count. From 79a7185b58db65d2ae9d44a605bbf0b64745bbf7 Mon Sep 17 00:00:00 2001 From: Jan Buchar Date: Fri, 21 Aug 2026 18:12:34 +0200 Subject: [PATCH 05/16] Do not ignore symbols that are in fact part of the public API, remove obsolete `@private` decorators --- docs/public-api/crawlee-types.api.md | 6 ++++++ docs/upgrading/upgrading_v4.md | 3 +-- packages/basic-crawler/src/internals/basic-crawler.ts | 1 - packages/browser-pool/src/browser-pool.ts | 1 - packages/core/src/storages/request_manager_tandem.ts | 2 -- packages/types/src/utility-types.ts | 2 -- 6 files changed, 7 insertions(+), 8 deletions(-) diff --git a/docs/public-api/crawlee-types.api.md b/docs/public-api/crawlee-types.api.md index 08a1a3a0ec3a..1e9f5c53b6f9 100644 --- a/docs/public-api/crawlee-types.api.md +++ b/docs/public-api/crawlee-types.api.md @@ -9,6 +9,9 @@ import type { Readable } from 'node:stream'; // @public (undocumented) export type AllowedHttpMethods = 'GET' | 'HEAD' | 'POST' | 'PUT' | 'DELETE' | 'TRACE' | 'OPTIONS' | 'CONNECT' | 'PATCH' | 'get' | 'head' | 'post' | 'put' | 'delete' | 'trace' | 'options' | 'connect' | 'patch'; +// @public (undocumented) +export type Awaitable = T | PromiseLike; + // @public export interface BaseHttpClient { sendRequest(request: Request, options?: SendRequestOptions): Promise; @@ -22,6 +25,9 @@ export interface BatchAddRequestsResult { unprocessedRequests: UnprocessedRequest[]; } +// @public (undocumented) +export type Constructor = new (...args: any[]) => T; + // @public (undocumented) export interface Cookie { domain?: string; diff --git a/docs/upgrading/upgrading_v4.md b/docs/upgrading/upgrading_v4.md index 229133f85b0c..42beff0bc528 100644 --- a/docs/upgrading/upgrading_v4.md +++ b/docs/upgrading/upgrading_v4.md @@ -321,7 +321,6 @@ Renamed options — pass `configuration` instead of `config`: - `Dataset.open()`, `KeyValueStore.open()` and `RequestQueue.open()` (`StorageOpenOptions`) - `useState()` (`UseStateOptions`) - `purgeDefaultStorages()` (both the options object and the legacy positional argument) -- `new Snapshotter()` (`SnapshotterOptions`) - `saveSnapshot()` in `@crawlee/playwright` and `@crawlee/puppeteer` (`SaveSnapshotOptions`) - `RecoverableStateOptions`, `RequestListOptions`, `CpuLoadSignalOptions` and `MemoryLoadSignalOptions` @@ -335,7 +334,7 @@ const store = await KeyValueStore.open(null, { config: new Configuration({ persi const store = await KeyValueStore.open(null, { configuration: new Configuration({ persistStorage: false }) }); ``` -Renamed properties — `Dataset.config`, `KeyValueStore.config`, `Snapshotter.config` and `BrowserLauncher.config` (including `PuppeteerLauncher`) are now `.configuration`. +Renamed properties — `Dataset.config`, `KeyValueStore.config` and `BrowserLauncher.config` (including `PuppeteerLauncher`) are now `.configuration`. The `configuration` crawler option is unchanged, as are `serviceLocator.getConfiguration()` and `serviceLocator.setConfiguration()`. diff --git a/packages/basic-crawler/src/internals/basic-crawler.ts b/packages/basic-crawler/src/internals/basic-crawler.ts index 7939f4dd9be3..c5c54433c13d 100644 --- a/packages/basic-crawler/src/internals/basic-crawler.ts +++ b/packages/basic-crawler/src/internals/basic-crawler.ts @@ -1938,7 +1938,6 @@ export class BasicCrawler< /** * Opens the default {@apilink RequestQueue} — the crawler's own, read from when the caller supplied nothing. - * @private */ private async openOwnedRequestQueue(): Promise { // The first crawler instance uses the default queue (null identifier); diff --git a/packages/browser-pool/src/browser-pool.ts b/packages/browser-pool/src/browser-pool.ts index 572efa011535..ad958e2c79fd 100644 --- a/packages/browser-pool/src/browser-pool.ts +++ b/packages/browser-pool/src/browser-pool.ts @@ -889,7 +889,6 @@ export class BrowserPool< /** * Picks plugins round robin. - * @private */ private pickBrowserPlugin() { const pluginIndex = this.#pageCounter % this.browserPlugins.length; diff --git a/packages/core/src/storages/request_manager_tandem.ts b/packages/core/src/storages/request_manager_tandem.ts index 1a5c4c607cb5..d1c3127ffed9 100644 --- a/packages/core/src/storages/request_manager_tandem.ts +++ b/packages/core/src/storages/request_manager_tandem.ts @@ -56,7 +56,6 @@ export class RequestManagerTandem implements IRequestManager { /** * Resolves the writable request manager, opening it lazily (via the factory) on first use and memoizing the result. - * @private */ private async getRequestManager(): Promise { if (this.#resolvedRequestManager === undefined) { @@ -79,7 +78,6 @@ export class RequestManagerTandem implements IRequestManager { * * @returns `true` if a request was successfully transferred (or there was nothing to transfer), and `false` if a * transfer was attempted but failed - in which case the caller should not fetch from the manager this round. - * @private */ private async transferNextRequestToQueue(): Promise { const request = await this.#requestLoader.fetchNextRequest(); diff --git a/packages/types/src/utility-types.ts b/packages/types/src/utility-types.ts index dc8c7dee487c..a9a61f5f3026 100644 --- a/packages/types/src/utility-types.ts +++ b/packages/types/src/utility-types.ts @@ -1,9 +1,7 @@ export type Dictionary = Record; -/** @ignore */ export type Constructor = new (...args: any[]) => T; -/** @ignore */ export type Awaitable = T | PromiseLike; export type AllowedHttpMethods = From 5d25c27aa4f7c2b881926eaa4469f4561c9431bf Mon Sep 17 00:00:00 2001 From: Jan Buchar Date: Thu, 27 Aug 2026 21:01:38 +0200 Subject: [PATCH 06/16] docs: Fix stale references left by the extractor move, and two broken examples MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `scripts/api-extractor/` moved out to `apify/api-extractor-report` in #4045, leaving `docs/public-api/README.md` pointing at a directory that no longer exists in three places. The exclusions are a `--exclude` flag now, and the generator's behaviour is pinned by that repository rather than this one, which is worth saying out loud. Also two examples that could not work: - `session_management.mdx` imported `ISessionPool` from `crawlee`, which does not export it — the whole custom-session-pool example failed to compile. It lives in `@crawlee/types`. - `BrowserPool.hasFreeBrowserSlot()` documented its cap as `BrowserPoolOptions.maxOpenBrowsers`, which does not exist; a reader who passed it would hit the `strictObject` validator. It is an internal field, `Infinity` unless `RemoteBrowserPool` sets it from its own option. --- docs/guides/session_management.mdx | 5 ++++- docs/public-api/README.md | 19 +++++++++++-------- packages/browser-pool/src/browser-pool.ts | 9 +++++++-- 3 files changed, 22 insertions(+), 11 deletions(-) diff --git a/docs/guides/session_management.mdx b/docs/guides/session_management.mdx index 8c11e35a9b88..fdbf982b502a 100644 --- a/docs/guides/session_management.mdx +++ b/docs/guides/session_management.mdx @@ -249,8 +249,11 @@ The `await using` syntax needs Node.js 24 or later. On Node.js 22 call `ISessionPool` interface as its `sessionPool` option, not just the built-in `SessionPool`. The contract is intentionally tiny — a single `getSession()` / `getSession(id)` method that hands out an `ISession` for a request. This lets you plug in a remote, shared, or database-backed session strategy without subclassing `SessionPool` or copying its internals. +`ISessionPool` and `ISession` are owned by `@crawlee/types`, which the `crawlee` meta-package does not re-export — add it to your dependencies to import them. + ```ts -import { BasicCrawler, Session, type ISessionPool } from 'crawlee'; +import { BasicCrawler, Session } from 'crawlee'; +import type { ISessionPool } from '@crawlee/types'; class MySessionPool implements ISessionPool { private readonly sessions = new Map(); diff --git a/docs/public-api/README.md b/docs/public-api/README.md index ea6e99ec9195..4c1aff3aeb64 100644 --- a/docs/public-api/README.md +++ b/docs/public-api/README.md @@ -61,8 +61,8 @@ reachable but unsupported. Both are legitimate; they answer different questions. - Mechanically, the reports are API Extractor's **`public`** variant, so `@internal` (`@alpha`/`@beta` too) is excluded. `@ignore` is a TypeDoc-era tag that API Extractor does - not act on, so the generator rewrites it to `@internal` before extraction — see `IGNORE_TAG` - in `scripts/api-extractor/run.ts`. It rewrites nothing else, which is worth knowing: + not act on, so the generator rewrites it to `@internal` before extraction. It rewrites + nothing else, which is worth knowing: **`@private` is inert here.** A member whose only tag is `@private` stays in the report and stays promised, so it is not a way to de-promise anything — use `@internal`. The generator stages the variant as `.public.api.md` under `temp/` and promotes it @@ -97,7 +97,8 @@ reachable but unsupported. Both are legitimate; they answer different questions. files) and is git-ignored. - `@crawlee/cli` and `@crawlee/templates` are deliberately excluded — they are tooling (a CLI binary and project scaffolding), not an importable API where we promise BC. The - exclude list lives in `scripts/api-extractor/run.ts`. + exclusions are passed as `--exclude` by the `api:check` / `api:extract` scripts in + `package.json`. - `crawlee.api.md` looks almost empty — eleven `export *` lines and no declarations — and that is correct rather than a gap. The meta-package re-exports; it declares nothing of its own. Everything it hands you belongs to a constituent package and is already inventoried there, so @@ -113,11 +114,13 @@ reachable but unsupported. Both are legitimate; they answer different questions. drop out of the barrel. Several hundred names are re-exported by more than one constituent today; every one of them is a single symbol reached by several paths, which is fine and raises nothing. -- The generator lives in `scripts/api-extractor/`. It temporarily strips the build's - injected `// @ts-ignore` comment lines from the `.d.ts` files (restoring them - afterwards) because API Extractor's AST walker trips over some of them; a small number - of packages additionally need a sanitized-mirror fallback. See the comments in - `scripts/api-extractor/run.ts` for details. +- The generator lives in its own repository, + [`apify/api-extractor-report`](https://github.com/apify/api-extractor-report), and runs via + `pnpm dlx` from the `api:check` / `api:extract` scripts — it is not vendored here, so its + behaviour is pinned by that repository rather than by this one. It temporarily strips the + build's injected `// @ts-ignore` comment lines from the `.d.ts` files (restoring them + afterwards) because API Extractor's AST walker trips over some of them; a small number of + packages additionally need a sanitized-mirror fallback. - **The reports are an inventory of what we promise, not a measure of how much code is reachable.** A symbol is in a report because we have committed to not breaking it; a symbol is absent because we have not. Absent does *not* mean inaccessible, and making it diff --git a/packages/browser-pool/src/browser-pool.ts b/packages/browser-pool/src/browser-pool.ts index ad958e2c79fd..b451a886d3cb 100644 --- a/packages/browser-pool/src/browser-pool.ts +++ b/packages/browser-pool/src/browser-pool.ts @@ -1028,8 +1028,13 @@ export class BrowserPool< } /** - * Returns `true` if the pool can accept a new browser launch without exceeding - * {@link BrowserPoolOptions.maxOpenBrowsers}. Counts starting, active, and retired browsers. + * Returns `true` if the pool can accept a new browser launch without exceeding `maxOpenBrowsers`. + * Counts starting, active, and retired browsers. + * + * A plain `BrowserPool` leaves `maxOpenBrowsers` at `Infinity`, so this only returns `false` when something + * has set a cap — {@apilink RemoteBrowserPool} does, from its own + * {@apilink RemoteBrowserPoolOptions.maxOpenBrowsers|`maxOpenBrowsers`} option. There is no + * `BrowserPoolOptions` key for it. * * @internal */ From 7dec8f12ca2e4eaf0ee907f8442ce4510bd194db Mon Sep 17 00:00:00 2001 From: Jan Buchar Date: Wed, 23 Sep 2026 16:58:01 +0200 Subject: [PATCH 07/16] Simplify public-api/README.md --- CONTRIBUTING.md | 13 ++++ docs/public-api/README.md | 134 +++----------------------------------- 2 files changed, 22 insertions(+), 125 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 3003bcfba79e..4bc8d366a83e 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -63,6 +63,19 @@ yay -S libffi7 icu66 libwebp052 flite-unpatched sudo ln -s /usr/lib/libpcre.so /usr/lib/libpcre.so.3 ``` +## Public API reports + +`docs/public-api/*.api.md` map what each package promises not to break (see [its README](docs/public-api/README.md)). After changing any package's public surface, regenerate and commit them: + +```sh +pnpm build # reports are generated from dist/ +pnpm api:extract +``` + +CI runs `pnpm api:check`, which fails when a committed report is out of date. Either the surface change is intended (commit the updated report so reviewers see the diff) or it is accidental (fix it). + +It also fails when an untagged signature references an `@internal` type, since the report would then use a type it never declares. Regenerating won't fix that: either drop the tag from the referenced type or keep it out of the public signature. + ## Testing in Crawlee with vitest There are a few small differences between how testing in jest and vitest works. Mostly, they relate to what to do, and not do anymore. diff --git a/docs/public-api/README.md b/docs/public-api/README.md index 4c1aff3aeb64..6368024aa5b4 100644 --- a/docs/public-api/README.md +++ b/docs/public-api/README.md @@ -1,131 +1,15 @@ # Public API surface maps -Each `*.api.md` file in this folder is a generated **map of the public, type-level -interface** of one publishable `@crawlee/*` package — every exported class, method, -property, function, and type, with full signatures. These reports define **where we -promise backwards compatibility**. +Each `*.api.md` file here is a generated map of the public, type-level interface of one `@crawlee/*` package. Being in a report is the backwards-compatibility promise; being absent means no promise, not that the symbol is unreachable. See the [Public API guide](../guides/public_api.mdx) for what that means for users. -They are produced by [API Extractor](https://api-extractor.com/) from the built -`dist/index.d.ts` of each package. +- **Untagged** members are promised. The codebase does not use explicit `@public` tags. +- **`@internal`** (and the legacy `@ignore`) members are trimmed from the report but stay exported and present in the `.d.ts`. Use `#private` / `private` when something should be unreachable, `@internal` when it should be reachable but unsupported. +- **`@private` is inert**: it does not remove anything from the report. -## What a tag means +The reports are generated by [`apify/api-extractor-report`](https://github.com/apify/api-extractor-report); its [README](https://github.com/apify/api-extractor-report#readme) explains how they are produced and what it fixes up in API Extractor's output. Regenerating and CI checks are covered in [CONTRIBUTING.md](https://github.com/apify/crawlee/blob/master/CONTRIBUTING.md#public-api-reports). -These reports are an **inventory of backwards-compatibility promises**. Membership is the -promise; the tags are how a member joins or leaves it. +## Reading a report -- **Untagged — promised.** In the report, and covered by backwards compatibility. The codebase - does not use explicit `@public` tags, so untagged is the default and anything you add is - promised unless you say otherwise. -- **`@internal` (and the legacy `@ignore`) — not promised.** Trimmed from the report. It is - still exported, still present in the published `.d.ts`, and still has working - auto-completion. You may use it; it may change or disappear in any release, including a - patch. - -**We deliberately do not strip these members from the `.d.ts`.** No `stripInternal`, no -parallel "internal build". Two reasons: crawlee's own packages consume plenty of each other's -internals, and an escape hatch that lets someone unblock themselves — knowingly, at their own -risk — is worth more than one we have taken away. So the tag is a statement about *support*, -not about *access*. - -The practical consequence when reviewing: tagging something `@internal` does not make it -harder to reach, it only stops us owing anyone stability on it. Reach for `#private` or TS -`private` when a member genuinely should be unreachable, and for `@internal` when it should be -reachable but unsupported. Both are legitimate; they answer different questions. - -## Workflow - -- After changing any package's public surface, regenerate the reports and commit them: - - ```sh - pnpm build # the reports are generated from dist/ - pnpm api:extract - ``` - -- CI runs `pnpm api:check`, which fails if a committed report is out of date. A failing - check means you changed the public API: either that change is intentional (commit the - updated report — reviewers will see the surface diff) or it was accidental (fix it). - -- `api:check` also fails if a report ends up referencing a symbol it never declares, which - leaves the committed map describing a type nothing in it defines. Regenerating cannot fix - that; it has to be fixed in the source. In practice it means a `@public` symbol's signature - references an `@internal`/`@ignore`-d one, so the referenced type is trimmed out from under - it. Either drop the referenced type's tag (it is reachable from the public API, so users can - already depend on it) or keep it out of the public signature. An untagged symbol is - implicitly public, which is the convention here — the codebase does not use explicit - `@public` tags. - - A symbol that is merely missing from the package's exports does **not** need fixing: see the - note on forgotten exports below. - -## Notes - -- Mechanically, the reports are API Extractor's **`public`** variant, so `@internal` - (`@alpha`/`@beta` too) is excluded. `@ignore` is a TypeDoc-era tag that API Extractor does - not act on, so the generator rewrites it to `@internal` before extraction. It rewrites - nothing else, which is worth knowing: - **`@private` is inert here.** A member whose only tag is `@private` stays in the report and - stays promised, so it is not a way to de-promise anything — use `@internal`. - The generator stages the variant as `.public.api.md` under `temp/` and promotes it - onto the committed `.api.md`, so the tracked filenames stay stable. -- API Extractor builds the import list before it trims the non-`@public` declarations and - never revisits it, so a type reachable only from an `@internal` member would linger as a - bare import and read as public surface. There is no config option for this, so the - generator post-processes each report: it parses the fenced TypeScript and drops imports - whose binding is referenced by no declaration that survived the trim. -- **Forgotten exports** — types the public API references but the entry point never exports — - are included in the report via `includeForgottenExports` and carry an explicit banner: - - ```ts - // Not exported by the entry point; reachable only as a referenced type. - // @public (undocumented) - interface SitemapUrlData { - ``` - - Their *shape* is part of the surface we promise not to break, but their *name* is not - importable, so they are emitted without `export`. API Extractor labels them `@public - (undocumented)` like anything else, which is indistinguishable from a real export at a - glance, hence the added banner. The alternative was exporting every such type from its - package — ~38 new public exports, committing us to names we never meant to publish. If you - *want* one importable, export it deliberately and the report will show it with `export`. -- Because API Extractor decides both of the above before the `@public` trim, it also offers - declarations for symbols reachable only from members that never reach the report. The - generator drops those the same way it drops dead imports, so the report carries nothing it - does not refer to. Only symbols flagged `ae-forgotten-export` are eligible, which is what - keeps genuinely reachable declarations (e.g. the `social` namespace in `@crawlee/utils`, - whose members are exposed through a `declare namespace` block) from being pruned. -- `docs/public-api/temp/` holds intermediate reports (including the staged `.public.api.md` - files) and is git-ignored. -- `@crawlee/cli` and `@crawlee/templates` are deliberately excluded — they are tooling - (a CLI binary and project scaffolding), not an importable API where we promise BC. The - exclusions are passed as `--exclude` by the `api:check` / `api:extract` scripts in - `package.json`. -- `crawlee.api.md` looks almost empty — eleven `export *` lines and no declarations — and that - is correct rather than a gap. The meta-package re-exports; it declares nothing of its own. - Everything it hands you belongs to a constituent package and is already inventoried there, so - a break in any of it shows up in that package's report first. Only two kinds of change are - meta-package-only, and this report catches both: dropping an `export *` line, and changing - something the meta-package declares itself — the `utils` bag that used to live here was in - the report, and its removal showed up as a diff. Listing the ~380 re-exported names instead - would restate promises where they are not made and churn on every constituent change. - - The case that looks like a hole is not one. If two constituents ever export *different* - symbols under the same name, TypeScript raises `TS2308` ("has already exported a member - named …") in `packages/crawlee/src/index.ts`, so the build fails — the name does not quietly - drop out of the barrel. Several hundred names are re-exported by more than one constituent - today; every one of them is a single symbol reached by several paths, which is fine and - raises nothing. -- The generator lives in its own repository, - [`apify/api-extractor-report`](https://github.com/apify/api-extractor-report), and runs via - `pnpm dlx` from the `api:check` / `api:extract` scripts — it is not vendored here, so its - behaviour is pinned by that repository rather than by this one. It temporarily strips the - build's injected `// @ts-ignore` comment lines from the `.d.ts` files (restoring them - afterwards) because API Extractor's AST walker trips over some of them; a small number of - packages additionally need a sanitized-mirror fallback. -- **The reports are an inventory of what we promise, not a measure of how much code is - reachable.** A symbol is in a report because we have committed to not breaking it; a symbol - is absent because we have not. Absent does *not* mean inaccessible, and making it - inaccessible is explicitly not the goal — see "What a tag means" above. Shrinking a report - is therefore only ever a *consequence* of deciding that something was never a promise, never - a target in its own right. A change that removes entries without changing any decision has - achieved nothing; a change that adds entries because we decided to support something is a - success. Issue #3109 tracks that decision-making, not a line count. +- **Forgotten exports** are types the public API references but the entry point does not export. They appear without `export` under a `// Not exported by the entry point` banner: their shape is promised, their name is not importable. Export one deliberately if it should be. +- **`crawlee.api.md` is nearly empty on purpose.** The meta-package only has `export *` lines, and everything they re-export is inventoried in the constituent package's report. +- `@crawlee/cli` and `@crawlee/templates` are excluded: they are tooling, not an importable API. From 00e3effeb3102efe7338be466d59865f7eb90e27 Mon Sep 17 00:00:00 2001 From: Jan Buchar Date: Wed, 23 Sep 2026 17:01:54 +0200 Subject: [PATCH 08/16] Address review feedback --- docs/guides/public_api.mdx | 2 +- docs/guides/session_management.mdx | 2 +- docs/upgrading/upgrading_v4.md | 1 + .../basic-crawler/src/internals/crawlers/crawler_commons.ts | 4 +--- 4 files changed, 4 insertions(+), 5 deletions(-) diff --git a/docs/guides/public_api.mdx b/docs/guides/public_api.mdx index ecb7e544c9b7..ed63efb24c11 100644 --- a/docs/guides/public_api.mdx +++ b/docs/guides/public_api.mdx @@ -36,7 +36,7 @@ import { someInternalHelper } from 'crawlee'; ``` If you find yourself reaching for an internal member to get something done, that is worth -telling us about — [open an issue](https://github.com/apify/crawlee/issues). Those reports are +telling us about, so you should [open an issue](https://github.com/apify/crawlee/issues). Those reports are what we use to decide which extension points deserve a real, supported API. ## Things that are not types diff --git a/docs/guides/session_management.mdx b/docs/guides/session_management.mdx index fdbf982b502a..eac1cb5276af 100644 --- a/docs/guides/session_management.mdx +++ b/docs/guides/session_management.mdx @@ -249,7 +249,7 @@ The `await using` syntax needs Node.js 24 or later. On Node.js 22 call `ISessionPool` interface as its `sessionPool` option, not just the built-in `SessionPool`. The contract is intentionally tiny — a single `getSession()` / `getSession(id)` method that hands out an `ISession` for a request. This lets you plug in a remote, shared, or database-backed session strategy without subclassing `SessionPool` or copying its internals. -`ISessionPool` and `ISession` are owned by `@crawlee/types`, which the `crawlee` meta-package does not re-export — add it to your dependencies to import them. +`ISessionPool` and `ISession` are owned by `@crawlee/types`, which the `crawlee` meta-package does not re-export. Add it to your dependencies to import them. ```ts import { BasicCrawler, Session } from 'crawlee'; diff --git a/docs/upgrading/upgrading_v4.md b/docs/upgrading/upgrading_v4.md index 230c184488d4..3412bd7e6519 100644 --- a/docs/upgrading/upgrading_v4.md +++ b/docs/upgrading/upgrading_v4.md @@ -2385,6 +2385,7 @@ The crawler-only parts of `@crawlee/core` moved to `@crawlee/basic`, so that `@c - the crawler-only error classes: `RetryRequestError`, `RequestThrottledError`, `PersistentRateLimitError`, `NavigationSkippedError`, `MissingSessionError`, `MissingRouteError`, `RequestHandlerError` and the `ContextPipeline*Error` types `@crawlee/basic` re-exports everything from `@crawlee/core`, so `import { SessionPool } from '@crawlee/basic'` (or from `crawlee`, `@crawlee/http`, `@crawlee/playwright`, …) keeps working unchanged. Only imports written against `@crawlee/core` itself need to be pointed at `@crawlee/basic`. + ### The `utils` bag is removed from the `crawlee` meta-package The `crawlee` meta-package exported a `utils` object — the last remnant of v2's `Apify.utils` namespace — bundling `utils.puppeteer`, `utils.playwright`, `utils.log`, `utils.social`, `utils.sleep`, `utils.downloadListOfUrls` and `utils.parseOpenGraph`. It is gone. Every member was already exported from `crawlee` under its own name, so the fix is to import that name directly: diff --git a/packages/basic-crawler/src/internals/crawlers/crawler_commons.ts b/packages/basic-crawler/src/internals/crawlers/crawler_commons.ts index d48af7828923..a8c4234fdd64 100644 --- a/packages/basic-crawler/src/internals/crawlers/crawler_commons.ts +++ b/packages/basic-crawler/src/internals/crawlers/crawler_commons.ts @@ -93,9 +93,7 @@ export type TypedContextEnqueueLinks< : EnqueueLinks; /** A {@apilink Request} that has been dispatched, so its `id` and `loadedUrl` are guaranteed to be present. */ -// `Required>` rather than an inline `{ [P in 'id' | 'loadedUrl']-?: R[P] }`: only a homomorphic -// mapped type (one keyed by `keyof X`, as `Required` is) strips `undefined` from the property type. Keyed -// by a literal union it would merely drop the `?`, leaving `string | undefined`. +// `Required` (homomorphic) strips `undefined`; an inline mapped type over a literal union would only drop the `?`. export type LoadedRequest = R & Required>; /** @internal */ From 4148e93711fe7de6348e4f22e6acaaab63f0fe15 Mon Sep 17 00:00:00 2001 From: Jan Buchar Date: Thu, 24 Sep 2026 15:22:03 +0200 Subject: [PATCH 09/16] Revert inlined RequestListSource --- docs/public-api/crawlee-core.api.md | 9 ++++++--- docs/upgrading/upgrading_v4.md | 2 +- packages/core/src/storages/request_list.ts | 11 ++++++----- 3 files changed, 13 insertions(+), 9 deletions(-) diff --git a/docs/public-api/crawlee-core.api.md b/docs/public-api/crawlee-core.api.md index d1ae8896e92a..8a23a5c11d1e 100644 --- a/docs/public-api/crawlee-core.api.md +++ b/docs/public-api/crawlee-core.api.md @@ -670,7 +670,7 @@ export class RequestList implements IRequestLoader { getTotalCount(): Promise; // (undocumented) markRequestAsHandled(request: Request_2): Promise; - static open(listNameOrOptions: string | null | RequestListOptions, sources?: (string | Source)[], options?: RequestListOptions): Promise; + static open(listNameOrOptions: string | null | RequestListOptions, sources?: RequestListSource[], options?: RequestListOptions): Promise; // (undocumented) persistState(): Promise; teardown(): Promise; @@ -684,13 +684,16 @@ export interface RequestListOptions { persistRequestsKey?: string; persistStateKey?: string; proxyConfiguration?: IProxyConfiguration; - sources?: (string | Source)[]; + sources?: RequestListSource[]; sourcesFunction?: RequestListSourcesFunction; state?: RequestListState; } // @public (undocumented) -export type RequestListSourcesFunction = () => Promise<(string | Source)[]>; +export type RequestListSource = string | Source; + +// @public (undocumented) +export type RequestListSourcesFunction = () => Promise; // @public export interface RequestListState { diff --git a/docs/upgrading/upgrading_v4.md b/docs/upgrading/upgrading_v4.md index 3412bd7e6519..a41103ab60a6 100644 --- a/docs/upgrading/upgrading_v4.md +++ b/docs/upgrading/upgrading_v4.md @@ -1802,7 +1802,7 @@ await enqueueLinks({ urls, requestManager }); #### Removed loader and manager type aliases -- `RequestListSource`, `UrlList` and `NewUrlOptions` are gone; the signatures that used them now spell their types out inline (`(string | Source)[]`, `(string | null)[]` and `{ request?: Request }` respectively). No behavioral change — replace the alias with the expansion if you referenced it. +- `UrlList` and `NewUrlOptions` are gone; the signatures that used them now spell their types out inline (`(string | null)[]` and `{ request?: Request }` respectively). No behavioral change — replace the alias with the expansion if you referenced it. - `RequestManagerOpener` is no longer exported, along with the `ThrottlingRequestManagerOptions.requestManagerOpener` option that took one. ## Only if you configure or implement storage backends diff --git a/packages/core/src/storages/request_list.ts b/packages/core/src/storages/request_list.ts index fdbdce217d82..c7e39930f376 100644 --- a/packages/core/src/storages/request_list.ts +++ b/packages/core/src/storages/request_list.ts @@ -161,7 +161,7 @@ export interface RequestListOptions { * ] * ``` */ - sources?: (string | Source)[]; + sources?: RequestListSource[]; /** * A function that will be called to get the sources for the `RequestList`, but only if `RequestList` @@ -391,7 +391,7 @@ export class RequestList implements IRequestLoader { #persistRequestsKey?: string; #store?: KeyValueStore; #keepDuplicateUrls: boolean; - #sources: (string | Source)[]; + #sources: RequestListSource[]; #sourcesFunction?: RequestListSourcesFunction; #proxyConfiguration?: IProxyConfiguration; #httpClient?: BaseHttpClient; @@ -742,7 +742,7 @@ export class RequestList implements IRequestLoader { * If the `source` parameter is a string or plain object and not an instance * of a `Request`, then the function creates a `Request` instance. */ - private addRequest(source: string | Source) { + private addRequest(source: RequestListSource) { let request: Request | RequestOptions; const type = typeof source; @@ -903,7 +903,7 @@ export class RequestList implements IRequestLoader { */ static async open( listNameOrOptions: string | null | RequestListOptions, - sources?: (string | Source)[], + sources?: RequestListSource[], options: RequestListOptions = {}, ): Promise { if (listNameOrOptions != null && typeof listNameOrOptions === 'object') { @@ -973,4 +973,5 @@ export interface RequestListState { inProgress: string[]; } -export type RequestListSourcesFunction = () => Promise<(string | Source)[]>; +export type RequestListSource = string | Source; +export type RequestListSourcesFunction = () => Promise; From 8e90da90ee7866b867adccf4da72a847cccd766f Mon Sep 17 00:00:00 2001 From: Jan Buchar Date: Thu, 24 Sep 2026 15:33:11 +0200 Subject: [PATCH 10/16] Fix test --- packages/http-client/src/base-http-client.ts | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/packages/http-client/src/base-http-client.ts b/packages/http-client/src/base-http-client.ts index fbf33956b443..9e8c18a58f4d 100644 --- a/packages/http-client/src/base-http-client.ts +++ b/packages/http-client/src/base-http-client.ts @@ -202,9 +202,7 @@ export abstract class BaseHttpClient implements BaseHttpClientInterface { while (true) { // Like `fetch`, set the jar cookies on a copy, so that the next redirect hop does not inherit them - await this.#applyCookies(new Request(currentRequest), cookieJar); - - const response = await this.fetch(currentRequest, { + const response = await this.fetch(await this.#applyCookies(new Request(currentRequest), cookieJar), { signal, proxyUrl, cookieJar, From 5e7d33190d2c4beed3e4764e79d25b8e12addd69 Mon Sep 17 00:00:00 2001 From: Jan Buchar Date: Fri, 25 Sep 2026 14:59:02 +0200 Subject: [PATCH 11/16] Do not expose internal utils --- docs/public-api/crawlee-basic.api.md | 2 +- .../basic-crawler/src/internals/enqueue_links/enqueue_links.ts | 2 +- packages/basic-crawler/src/internals/sitemap_request_loader.ts | 3 ++- 3 files changed, 4 insertions(+), 3 deletions(-) diff --git a/docs/public-api/crawlee-basic.api.md b/docs/public-api/crawlee-basic.api.md index a2945edcc63f..c6ad3cd0b44b 100644 --- a/docs/public-api/crawlee-basic.api.md +++ b/docs/public-api/crawlee-basic.api.md @@ -15,7 +15,7 @@ import { CriticalError } from '@crawlee/core'; import { Dataset } from '@crawlee/core'; import type { DatasetExportOptions } from '@crawlee/core'; import { Dictionary } from '@crawlee/types'; -import { EnqueueStrategy } from '@crawlee/utils/internal'; +import { EnqueueStrategy } from '@crawlee/utils'; import type { EnqueueStrategyOption } from '@crawlee/core'; import { EventManager } from '@crawlee/core'; import type { HttpRequestOptions } from '@crawlee/types'; diff --git a/packages/basic-crawler/src/internals/enqueue_links/enqueue_links.ts b/packages/basic-crawler/src/internals/enqueue_links/enqueue_links.ts index ce95352617f2..1b146ba644ed 100644 --- a/packages/basic-crawler/src/internals/enqueue_links/enqueue_links.ts +++ b/packages/basic-crawler/src/internals/enqueue_links/enqueue_links.ts @@ -1,5 +1,5 @@ import type { Dictionary } from '@crawlee/types'; -import { EnqueueStrategy } from '@crawlee/utils/internal'; +import { EnqueueStrategy } from '@crawlee/utils'; import { getDomain } from 'tldts'; import type { EnqueueStrategyOption, RequestQueueOperationOptions } from '@crawlee/core'; diff --git a/packages/basic-crawler/src/internals/sitemap_request_loader.ts b/packages/basic-crawler/src/internals/sitemap_request_loader.ts index 94f1bada133c..f46ef7ae62ff 100644 --- a/packages/basic-crawler/src/internals/sitemap_request_loader.ts +++ b/packages/basic-crawler/src/internals/sitemap_request_loader.ts @@ -2,7 +2,8 @@ import { Transform } from 'node:stream'; import type { BaseHttpClient } from '@crawlee/http-client'; import type { ParseSitemapOptions } from '@crawlee/utils'; -import { EnqueueStrategy, parseArgument, parseSitemap, schemas } from '@crawlee/utils/internal'; +import { EnqueueStrategy } from '@crawlee/utils'; +import { parseArgument, parseSitemap, schemas } from '@crawlee/utils/internal'; import { minimatch } from 'minimatch'; import type { RequiredDeep } from 'type-fest'; import { z } from 'zod'; From 51b0101ef23bd9ff6336fa012c876fdb993b8e28 Mon Sep 17 00:00:00 2001 From: Jan Buchar Date: Fri, 25 Sep 2026 15:41:31 +0200 Subject: [PATCH 12/16] Remove docs that suggested you should extend context in FileDownload --- docs/upgrading/upgrading_v4.md | 4 ++-- .../http-crawler/src/internals/file-download.ts | 13 ------------- 2 files changed, 2 insertions(+), 15 deletions(-) diff --git a/docs/upgrading/upgrading_v4.md b/docs/upgrading/upgrading_v4.md index a41103ab60a6..be5ab920a079 100644 --- a/docs/upgrading/upgrading_v4.md +++ b/docs/upgrading/upgrading_v4.md @@ -974,9 +974,9 @@ These are still present at runtime, but they are excluded from the documented su ### `HttpCrawler`, `FileDownload` and `JSDOMCrawler` internals are no longer accessible to subclasses -`HttpCrawler.isRequestBlocked` and `FileDownload.buildContextPipeline` are now `private`. Block detection is configured through `retryOnBlocked` and `blockedStatusCodes`; to add your own checks, throw a `SessionError` from a `postNavigationHook`. To extend the `FileDownload` context pipeline, pass a `contextPipelineBuilder` to the constructor rather than subclassing. +`HttpCrawler.isRequestBlocked` and `FileDownload.buildContextPipeline` are now `private`. Block detection is configured through `retryOnBlocked` and `blockedStatusCodes`; to add your own checks, throw a `SessionError` from a `postNavigationHook`. -`HttpCrawler.buildContextPipeline` and `BrowserCrawler.buildContextPipeline` stay `protected` and overridable, and are supported extension points: they are the two levels that genuinely compose context stages, so override one of them to add your own. `BasicCrawler.buildContextPipeline` is gone — it only ever returned an empty pipeline — and the builders on `CheerioCrawler`, `JSDOMCrawler`, `LinkeDOMCrawler`, `PlaywrightCrawler`, `AdaptivePlaywrightCrawler`, `PuppeteerCrawler` and `StagehandCrawler` are now `#private`. If you were overriding one of those, pass a `contextPipelineBuilder` to the constructor instead. +`HttpCrawler.buildContextPipeline` and `BrowserCrawler.buildContextPipeline` stay `protected` and overridable, and are supported extension points: they are the two levels that genuinely compose context stages, so override one of them to add your own. `BasicCrawler.buildContextPipeline` is gone — it only ever returned an empty pipeline — and the builders on `CheerioCrawler`, `JSDOMCrawler`, `LinkeDOMCrawler`, `PlaywrightCrawler`, `AdaptivePlaywrightCrawler`, `PuppeteerCrawler` and `StagehandCrawler` are now `#private`. If you were overriding one of those to add context members, use the `extendContext` option instead. It runs before navigation, so it cannot see navigation-dependent members such as `page` or `$`; read those in a `postNavigationHook` or the `requestHandler`. `HttpCrawler.getNavigationTimeoutMillis` and `HttpCrawler.createDefaultConcurrencySystem` stay `protected` and overridable, but are marked `@internal` — they are implementation seams shared with `@crawlee/cheerio`, `@crawlee/jsdom` and `@crawlee/linkedom`, and their signatures may change in a minor release. diff --git a/packages/http-crawler/src/internals/file-download.ts b/packages/http-crawler/src/internals/file-download.ts index 4d460b702895..475a2100881d 100644 --- a/packages/http-crawler/src/internals/file-download.ts +++ b/packages/http-crawler/src/internals/file-download.ts @@ -53,19 +53,6 @@ export type FileDownloadRequestHandler< * * The crawler finishes when there are no more {@apilink Request} objects to crawl. * - * `FileDownload` has no `preNavigationHooks` / `postNavigationHooks` options. To adjust the crawling context before - * the request is made, pass your own {@apilink BasicCrawlerOptions.contextPipelineBuilder|`contextPipelineBuilder`}: - * - * ```ts - * const crawler = new FileDownload({ - * contextPipelineBuilder: () => - * ContextPipeline.create().compose(async () => ({ myField: 123 })), - * requestHandler({ myField }) { - * // ... - * }, - * }); - * ``` - * * New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's {@apilink ConcurrencySystem}. Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the `FileCrawler` constructor, or, for finer control, by injecting a pre-configured {@apilink ConcurrencySystem|`concurrencySystem`}. * * ## Example usage From ac1baa571bd882fd0e874408633e0be7634e74ca Mon Sep 17 00:00:00 2001 From: Jan Buchar Date: Fri, 25 Sep 2026 17:04:58 +0200 Subject: [PATCH 13/16] Do not accept contextPipelineBuilder in concrete crawler classes --- docs/public-api/crawlee-browser.api.md | 2 +- docs/public-api/crawlee-cheerio.api.md | 2 +- docs/public-api/crawlee-http.api.md | 4 +-- docs/public-api/crawlee-playwright.api.md | 2 +- .../src/internals/browser-crawler.ts | 4 +-- .../src/internals/cheerio-crawler.ts | 15 +++++---- .../http-crawler/src/internals/dom-crawler.ts | 16 +++------ .../src/internals/file-download.ts | 4 +-- .../src/internals/http-crawler.ts | 2 +- .../internals/adaptive-playwright-crawler.ts | 5 ++- .../src/internals/playwright-crawler.ts | 5 ++- .../src/internals/puppeteer-crawler.ts | 11 ++----- .../src/internals/stagehand-crawler.ts | 11 ++----- test/core/crawlers/file_download.test.ts | 33 +------------------ 14 files changed, 33 insertions(+), 83 deletions(-) diff --git a/docs/public-api/crawlee-browser.api.md b/docs/public-api/crawlee-browser.api.md index 62cd1c53a412..57b6a61d0b53 100644 --- a/docs/public-api/crawlee-browser.api.md +++ b/docs/public-api/crawlee-browser.api.md @@ -49,7 +49,7 @@ export abstract class BrowserCrawler = BrowserCrawlingContext, ContextExtension = Dictionary, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends object = {}> extends Omit, 'requestHandler' | 'failedRequestHandler' | 'errorHandler'> { +export interface BrowserCrawlerOptions = BrowserCrawlingContext, ContextExtension = Dictionary, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends object = {}> extends Omit, 'requestHandler' | 'failedRequestHandler' | 'errorHandler' | 'contextPipelineBuilder'> { browserPool?: IBrowserPool; errorHandler?: ErrorHandler; failedRequestHandler?: ErrorHandler; diff --git a/docs/public-api/crawlee-cheerio.api.md b/docs/public-api/crawlee-cheerio.api.md index 1e124e2da2e3..01c91e6e92d1 100644 --- a/docs/public-api/crawlee-cheerio.api.md +++ b/docs/public-api/crawlee-cheerio.api.md @@ -27,7 +27,7 @@ export class CheerioCrawler, ExtendedContex // @public (undocumented) export interface CheerioCrawlerOptions, ExtendedContext extends CheerioCrawlingContext = CheerioCrawlingContext & ContextExtension, UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler -Routes extends Record = Record, StatisticStateExtension extends object = {}> extends HttpCrawlerOptions, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> { +Routes extends Record = Record, StatisticStateExtension extends object = {}> extends Omit, ContextExtension, ExtendedContext, Routes, StatisticStateExtension>, 'contextPipelineBuilder'> { } // @public (undocumented) diff --git a/docs/public-api/crawlee-http.api.md b/docs/public-api/crawlee-http.api.md index a10914e9d546..40dd5085d0ba 100644 --- a/docs/public-api/crawlee-http.api.md +++ b/docs/public-api/crawlee-http.api.md @@ -60,7 +60,7 @@ export class DOMCrawler, ExtendedContext extends DOMCrawlingContext = DOMCrawlingContext & ContextExtension, Routes extends Record = Record, StatisticStateExtension extends object = {}> extends HttpCrawlerOptions, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> { +export interface DOMCrawlerOptions, ExtendedContext extends DOMCrawlingContext = DOMCrawlingContext & ContextExtension, Routes extends Record = Record, StatisticStateExtension extends object = {}> extends Omit, ContextExtension, ExtendedContext, Routes, StatisticStateExtension>, 'contextPipelineBuilder'> { parser: DOMParser_2; } @@ -97,7 +97,7 @@ export interface DOMParseResult { // @public export class FileDownload extends BasicCrawler { - constructor(options?: BasicCrawlerOptions); + constructor(options?: Omit, 'contextPipelineBuilder'>); } // @public (undocumented) diff --git a/docs/public-api/crawlee-playwright.api.md b/docs/public-api/crawlee-playwright.api.md index 057892ffc8cc..35b9801f8a93 100644 --- a/docs/public-api/crawlee-playwright.api.md +++ b/docs/public-api/crawlee-playwright.api.md @@ -97,7 +97,7 @@ export interface AdaptivePlaywrightCrawlerContext, ExtendedContext extends AdaptivePlaywrightCrawlerContext = AdaptivePlaywrightCrawlerContext & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends AdaptivePlaywrightCrawlerStatisticState = AdaptivePlaywrightCrawlerStatisticState> extends Omit, 'preNavigationHooks' | 'postNavigationHooks'>, Pick { +export interface AdaptivePlaywrightCrawlerOptions, ExtendedContext extends AdaptivePlaywrightCrawlerContext = AdaptivePlaywrightCrawlerContext & ContextExtension, Routes extends Record = Record>, StatisticStateExtension extends AdaptivePlaywrightCrawlerStatisticState = AdaptivePlaywrightCrawlerStatisticState> extends Omit, 'preNavigationHooks' | 'postNavigationHooks' | 'contextPipelineBuilder'>, Pick { postNavigationHooks?: AdaptivePostNavigationHook[]; preNavigationHooks?: AdaptiveHook[]; renderingTypeDetectionRatio?: number; diff --git a/packages/browser-crawler/src/internals/browser-crawler.ts b/packages/browser-crawler/src/internals/browser-crawler.ts index 9355ed96e988..e13ac47a0552 100644 --- a/packages/browser-crawler/src/internals/browser-crawler.ts +++ b/packages/browser-crawler/src/internals/browser-crawler.ts @@ -119,8 +119,8 @@ export interface BrowserCrawlerOptions< StatisticStateExtension extends object = {}, > extends Omit< BasicCrawlerOptions, - // Overridden with browser context - 'requestHandler' | 'failedRequestHandler' | 'errorHandler' + // Overridden with browser context; `contextPipelineBuilder` is supplied by the concrete crawler + 'requestHandler' | 'failedRequestHandler' | 'errorHandler' | 'contextPipelineBuilder' > { /** * The browser pool the crawler should serve its pages from. This is the single way to run a pool with diff --git a/packages/cheerio-crawler/src/internals/cheerio-crawler.ts b/packages/cheerio-crawler/src/internals/cheerio-crawler.ts index 16d81fcac6f8..c5c64fb8dcca 100644 --- a/packages/cheerio-crawler/src/internals/cheerio-crawler.ts +++ b/packages/cheerio-crawler/src/internals/cheerio-crawler.ts @@ -30,12 +30,15 @@ export interface CheerioCrawlerOptions< JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler Routes extends Record = Record, StatisticStateExtension extends object = {}, -> extends HttpCrawlerOptions< - CheerioCrawlingContext, - ContextExtension, - ExtendedContext, - Routes, - StatisticStateExtension +> extends Omit< + HttpCrawlerOptions< + CheerioCrawlingContext, + ContextExtension, + ExtendedContext, + Routes, + StatisticStateExtension + >, + 'contextPipelineBuilder' > {} export type CheerioHook< diff --git a/packages/http-crawler/src/internals/dom-crawler.ts b/packages/http-crawler/src/internals/dom-crawler.ts index ef5346bca72d..5bbb951c2dec 100644 --- a/packages/http-crawler/src/internals/dom-crawler.ts +++ b/packages/http-crawler/src/internals/dom-crawler.ts @@ -142,12 +142,9 @@ export interface DOMCrawlerOptions< ExtendedContext extends DOMCrawlingContext = DOMCrawlingContext & ContextExtension, Routes extends Record = Record, StatisticStateExtension extends object = {}, -> extends HttpCrawlerOptions< - DOMCrawlingContext, - ContextExtension, - ExtendedContext, - Routes, - StatisticStateExtension +> extends Omit< + HttpCrawlerOptions, ContextExtension, ExtendedContext, Routes, StatisticStateExtension>, + 'contextPipelineBuilder' > { /** * The DOM implementation to parse the response bodies with. Its parse result becomes part of the crawling @@ -180,12 +177,9 @@ export class DOMCrawler< constructor( options: DOMCrawlerOptions, ) { - const { parser, contextPipelineBuilder, ...rest } = options; + const { parser, ...rest } = options; - super({ - ...rest, - contextPipelineBuilder: contextPipelineBuilder ?? (() => this.buildContextPipeline()), - }); + super({ ...rest, contextPipelineBuilder: () => this.buildContextPipeline() }); this.#parser = parser; } diff --git a/packages/http-crawler/src/internals/file-download.ts b/packages/http-crawler/src/internals/file-download.ts index 475a2100881d..1d18f0216583 100644 --- a/packages/http-crawler/src/internals/file-download.ts +++ b/packages/http-crawler/src/internals/file-download.ts @@ -73,10 +73,10 @@ export type FileDownloadRequestHandler< */ export class FileDownload extends BasicCrawler { // TODO hooks - constructor(options: BasicCrawlerOptions = {}) { + constructor(options: Omit, 'contextPipelineBuilder'> = {}) { super({ ...options, - contextPipelineBuilder: options.contextPipelineBuilder ?? (() => this.#buildContextPipeline()), + contextPipelineBuilder: () => this.#buildContextPipeline(), }); } diff --git a/packages/http-crawler/src/internals/http-crawler.ts b/packages/http-crawler/src/internals/http-crawler.ts index c752726964f7..607f1c77a915 100644 --- a/packages/http-crawler/src/internals/http-crawler.ts +++ b/packages/http-crawler/src/internals/http-crawler.ts @@ -11,8 +11,8 @@ import type { GetUserDataFromRequest, LoadedRequest, CrawlingRequest, - RequestHandler, RequireContextPipeline, + RequestHandler, RouterHandler, RouterRoutes, RouteSchemas, diff --git a/packages/playwright-crawler/src/internals/adaptive-playwright-crawler.ts b/packages/playwright-crawler/src/internals/adaptive-playwright-crawler.ts index f0ed3f16024a..6646551190b9 100644 --- a/packages/playwright-crawler/src/internals/adaptive-playwright-crawler.ts +++ b/packages/playwright-crawler/src/internals/adaptive-playwright-crawler.ts @@ -179,7 +179,7 @@ export interface AdaptivePlaywrightCrawlerOptions< Routes, StatisticStateExtension >, - 'preNavigationHooks' | 'postNavigationHooks' + 'preNavigationHooks' | 'postNavigationHooks' | 'contextPipelineBuilder' >, Pick { /** @@ -367,7 +367,6 @@ export class AdaptivePlaywrightCrawler< preNavigationHooks = [], postNavigationHooks = [], extendContext, - contextPipelineBuilder, transactionalStorage, launchContext, headless, @@ -426,7 +425,7 @@ export class AdaptivePlaywrightCrawler< stateExtension: adaptivePlaywrightCrawlerStatisticState as StatisticStateExtensionOptions, }), - contextPipelineBuilder: contextPipelineBuilder ?? (() => this.#buildContextPipeline()), + contextPipelineBuilder: () => this.#buildContextPipeline(), // The base crawler must not wrap requests in a transaction of its own - this crawler opens // one per request handler attempt in `crawlOne` instead, forwarding the write policy of the // user-facing option (validated above) to those. diff --git a/packages/playwright-crawler/src/internals/playwright-crawler.ts b/packages/playwright-crawler/src/internals/playwright-crawler.ts index ec7f277630dc..18e4b27ff17f 100644 --- a/packages/playwright-crawler/src/internals/playwright-crawler.ts +++ b/packages/playwright-crawler/src/internals/playwright-crawler.ts @@ -224,8 +224,7 @@ export class PlaywrightCrawler< ) { const parsedOptions = parseArgument(options, PlaywrightCrawler.optionsSchema, 'PlaywrightCrawlerOptions'); - const { launchContext, headless, configuration, contextPipelineBuilder, ...browserCrawlerOptions } = - parsedOptions; + const { launchContext, headless, configuration, ...browserCrawlerOptions } = parsedOptions; if (launchContext.proxyUrl) { throw new Error( @@ -254,7 +253,7 @@ export class PlaywrightCrawler< remoteBrowser ? remotePlaywrightBrowserPool({ ...remoteBrowser, launchContext, headless, configuration }) : playwrightBrowserPool({ launchContext, headless, configuration }), - contextPipelineBuilder: contextPipelineBuilder ?? (() => this.#buildContextPipeline()), + contextPipelineBuilder: () => this.#buildContextPipeline(), }); } diff --git a/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts b/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts index bfa31ee084a3..7e4cc141e6cc 100644 --- a/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts +++ b/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts @@ -217,14 +217,7 @@ export class PuppeteerCrawler< ) { const parsedOptions = parseArgument(options, PuppeteerCrawler.optionsSchema, 'PuppeteerCrawlerOptions'); - const { - launchContext, - headless, - configuration, - proxyConfiguration, - contextPipelineBuilder, - ...browserCrawlerOptions - } = parsedOptions; + const { launchContext, headless, configuration, proxyConfiguration, ...browserCrawlerOptions } = parsedOptions; if (launchContext.proxyUrl) { throw new Error( @@ -257,7 +250,7 @@ export class PuppeteerCrawler< remoteBrowser ? remotePuppeteerBrowserPool({ ...remoteBrowser, launchContext, headless, configuration }) : puppeteerBrowserPool({ launchContext, headless, configuration }), - contextPipelineBuilder: contextPipelineBuilder ?? (() => this.#buildContextPipeline()), + contextPipelineBuilder: () => this.#buildContextPipeline(), }); } diff --git a/packages/stagehand-crawler/src/internals/stagehand-crawler.ts b/packages/stagehand-crawler/src/internals/stagehand-crawler.ts index d48bbc45c36c..e5a8d8de614a 100644 --- a/packages/stagehand-crawler/src/internals/stagehand-crawler.ts +++ b/packages/stagehand-crawler/src/internals/stagehand-crawler.ts @@ -428,14 +428,7 @@ export class StagehandCrawler< ) { const parsedOptions = parseArgument(options, StagehandCrawler.optionsSchema, 'StagehandCrawlerOptions'); - const { - stagehandOptions, - launchContext, - headless, - configuration, - contextPipelineBuilder, - ...browserCrawlerOptions - } = parsedOptions; + const { stagehandOptions, launchContext, headless, configuration, ...browserCrawlerOptions } = parsedOptions; if (options.browserPool) { // The raw options, not the parsed ones: `launchContext` has a default, so by now it is always set. @@ -471,7 +464,7 @@ export class StagehandCrawler< headless, configuration, })) as unknown as OwnedBrowserPool, - contextPipelineBuilder: contextPipelineBuilder ?? (() => this.#buildContextPipeline()), + contextPipelineBuilder: () => this.#buildContextPipeline(), }); } diff --git a/test/core/crawlers/file_download.test.ts b/test/core/crawlers/file_download.test.ts index 03046f6c426e..06751a99606d 100644 --- a/test/core/crawlers/file_download.test.ts +++ b/test/core/crawlers/file_download.test.ts @@ -5,8 +5,7 @@ import { pipeline } from 'node:stream/promises'; import { ReadableStream } from 'node:stream/web'; import { setTimeout } from 'node:timers/promises'; -import type { CrawlingContext, CrawlingRequest, LoadedRequest } from '@crawlee/http'; -import { ContextPipeline, FileDownload } from '@crawlee/http'; +import { FileDownload } from '@crawlee/http'; import { FetchHttpClient } from '@crawlee/http-client'; import express from 'express'; import { startExpressAppPromise } from '../../shared/_helper.js'; @@ -201,33 +200,3 @@ test('crawler waits for the stream to be consumed', async () => { expect(bufferedData.length).toBe(5 * 1024); expect(bufferedData).toEqual(await ReadableStreamGenerator.getUint8Array(5 * 1024, 789)); }); - -test('honours a user-supplied contextPipelineBuilder', async () => { - let builderCalls = 0; - const bodies: string[] = []; - - const crawler = new FileDownload({ - maxRequestRetries: 0, - contextPipelineBuilder: () => { - builderCalls++; - - return ContextPipeline.create().compose(async (context) => ({ - request: context.request as LoadedRequest, - response: new Response('stubbed'), - contentType: { type: 'text/plain', encoding: 'utf8' as BufferEncoding }, - })); - }, - requestHandler: async ({ response }) => { - bodies.push(await response.text()); - }, - }); - - const fileUrl = new URL('/file?size=1024&seed=123', url).toString(); - - const stats = await crawler.run([fileUrl]); - - expect(stats.requestsFailed).toBe(0); - expect(builderCalls).toBe(1); - // The supplied pipeline replaces the built-in download, so the handler sees the stub, not the file. - expect(bodies).toEqual(['stubbed']); -}); From d36dffeb5442ed0964f02e0e866f7310e308e47c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Martin=20Ad=C3=A1mek?= Date: Tue, 29 Sep 2026 11:10:50 +0200 Subject: [PATCH 14/16] docs: fix nits in the upgrading guide and JSDoc Correct the `@internal` list and utils entry-point notes, list `utils.enqueueLinks` in the removed bag, export `SitemapUrl` from `@crawlee/utils/internal` as documented, add the public API guide to the sidebar, and tidy comment formatting. --- docs/upgrading/upgrading_v4.md | 6 +++--- packages/basic-crawler/src/internals/enqueue_links/index.ts | 3 +-- .../basic-crawler/src/internals/session_pool/session.ts | 2 +- packages/browser-pool/src/browser-pool.ts | 3 +-- packages/core/src/proxy_configuration.ts | 1 + packages/utils/src/internal.ts | 2 +- website/sidebars.js | 1 + 7 files changed, 9 insertions(+), 9 deletions(-) diff --git a/docs/upgrading/upgrading_v4.md b/docs/upgrading/upgrading_v4.md index be5ab920a079..06d13590ecfb 100644 --- a/docs/upgrading/upgrading_v4.md +++ b/docs/upgrading/upgrading_v4.md @@ -968,7 +968,7 @@ The `running` and `hasFinishedBefore` flags on `BasicCrawler` were internal run- ### Declarations that are now `@internal` -These are still present at runtime, but they are excluded from the documented surface and can change without a major version bump: `Session.getState()`, `SessionOptions.log`, `Router.getTimeoutSecs()`, `Router.getMaxTimeoutSecs()`, `Request.skippedReason`, `RestrictedCrawlingContext.id`, `ThrottlingRequestManager.drop()`, `ThrottlingRequestManager.innerManager`, `ContextPipelineInitializationError`, `ContextPipelineCleanupError` and `RequestHandlerError`. +These are still present at runtime, but they are excluded from the documented surface and can change without a major version bump: `Session.getState()`, `SessionOptions.log`, `Router.getTimeoutSecs()`, `Router.getMaxTimeoutSecs()`, `Request.skippedReason`, `RestrictedCrawlingContext.id`, `ThrottlingRequestManager.drop()` and `ThrottlingRequestManager.innerManager`. `BLOCKED_STATUS_CODES` is now typed `readonly number[]`. Copy it (`[...BLOCKED_STATUS_CODES]`) if you were mutating it. @@ -2339,7 +2339,7 @@ Besides the resource-detection helpers above, several other `@crawlee/utils` exp - **Split into public and `/internal` entry points:** the main `@crawlee/utils` entry now exposes only the user-facing helpers (`sleep`, `htmlToText`, `extractUrls`, `downloadListOfUrls`, the `social` namespace, the Open Graph parser, and the robots/sitemap utilities `RobotsTxtFile`, `Sitemap` and `discoverValidSitemaps`). Helpers that primarily serve the crawler packages - e.g. `URL_NO_COMMAS_REGEX`, `URL_WITH_COMMAS_REGEX`, `extractUrlsFromCheerio`, `tryAbsoluteURL`, `expandShadowRoots`, and the blocked-detection and iterable helpers - moved to the `@crawlee/utils/internal` entry point, which carries no semver guarantees. They keep working, but imports need updating: `import { URL_NO_COMMAS_REGEX } from '@crawlee/utils/internal'`. - **Removed `CheerioRoot` and the cheerio type re-exports:** `CheerioRoot` was an alias for cheerio's own `CheerioAPI` and is gone; `parseWithCheerio()` and `htmlToText()` are typed with `CheerioAPI` directly. The crawler packages also no longer re-export `Cheerio`, `CheerioAPI` and `Element`, so `import type { CheerioAPI } from 'crawlee'` (or from `@crawlee/basic` / `@crawlee/puppeteer` / ...) breaks - import them from `cheerio` and `domhandler`, which are the packages that own them. - **`@crawlee/core` no longer re-exports the internal helpers:** `parseArgument`, `schemas` and `tryAbsoluteURL` reached `@crawlee/core` (and through it `@crawlee/basic`, `@crawlee/http`, `@crawlee/browser` and `crawlee`) as public exports, which put symbols from the no-semver `/internal` entry point back into a semver-stable surface. Import them from `@crawlee/utils/internal` instead. `ArgumentValidationError` is unaffected and stays exported from `@crawlee/core`. -- **`parseSitemap`, `parseArgument` and `expandShadowRoots` are no longer on the main entry:** `parseSitemap()` (together with the `SitemapUrl` type) moved to `@crawlee/utils/internal`; use the documented `Sitemap.load()` / `Sitemap.fromXmlString()` / `Sitemap.tryCommonNames()` statics, or `discoverValidSitemaps()`, which stay on `@crawlee/utils`. `parseArgument()` was reaching the main entry through a wildcard re-export and is now only on `@crawlee/utils/internal` (`ArgumentValidationError` is unaffected and stays public). `expandShadowRoots()` moved there too — it is a DOM function that is serialized into a browser page, not a Node helper. Because the `crawlee` meta-package re-exports `@crawlee/utils` wholesale, `import { parseSitemap } from 'crawlee'` (and the same for `parseArgument` / `expandShadowRoots`) breaks as well. +- **`parseSitemap` and `expandShadowRoots` are no longer on the main entry:** `parseSitemap()` (together with the `SitemapUrl` type) moved to `@crawlee/utils/internal`; use the documented `Sitemap.load()` / `Sitemap.fromXmlString()` / `Sitemap.tryCommonNames()` statics, or `discoverValidSitemaps()`, which stay on `@crawlee/utils`. `expandShadowRoots()` moved there too — it is a DOM function that is serialized into a browser page, not a Node helper. Because the `crawlee` meta-package re-exports `@crawlee/utils` wholesale, `import { parseSitemap } from 'crawlee'` (and the same for `expandShadowRoots`) breaks as well. #### `RobotsTxtFile.find` signature changed; sitemap options removed @@ -2388,7 +2388,7 @@ The crawler-only parts of `@crawlee/core` moved to `@crawlee/basic`, so that `@c ### The `utils` bag is removed from the `crawlee` meta-package -The `crawlee` meta-package exported a `utils` object — the last remnant of v2's `Apify.utils` namespace — bundling `utils.puppeteer`, `utils.playwright`, `utils.log`, `utils.social`, `utils.sleep`, `utils.downloadListOfUrls` and `utils.parseOpenGraph`. It is gone. Every member was already exported from `crawlee` under its own name, so the fix is to import that name directly: +The `crawlee` meta-package exported a `utils` object — the last remnant of v2's `Apify.utils` namespace — bundling `utils.puppeteer`, `utils.playwright`, `utils.log`, `utils.enqueueLinks`, `utils.social`, `utils.sleep`, `utils.downloadListOfUrls` and `utils.parseOpenGraph`. It is gone. Every member was already exported from `crawlee` under its own name, so the fix is to import that name directly: **Before:** ```typescript diff --git a/packages/basic-crawler/src/internals/enqueue_links/index.ts b/packages/basic-crawler/src/internals/enqueue_links/index.ts index 867f10e2eca4..14bd4ecf23be 100644 --- a/packages/basic-crawler/src/internals/enqueue_links/index.ts +++ b/packages/basic-crawler/src/internals/enqueue_links/index.ts @@ -1,6 +1,5 @@ export * from './enqueue_links.js'; -// Not `export *`: `UrlPatternObject` is the compiled internal form of a `UrlPatternInput` and carries no semver -// guarantees. Internal consumers import it from `./shared.js` directly. +// Not `export *`: keeps the internal `UrlPatternObject` off the public surface. export { applyRequestTransform, constructGlobObjectsFromGlobs, diff --git a/packages/basic-crawler/src/internals/session_pool/session.ts b/packages/basic-crawler/src/internals/session_pool/session.ts index f27412fda272..a87d6ecb38b0 100644 --- a/packages/basic-crawler/src/internals/session_pool/session.ts +++ b/packages/basic-crawler/src/internals/session_pool/session.ts @@ -265,7 +265,7 @@ export class Session implements ISession { /** * Gets session state for persistence in KeyValueStore. - + * * @internal */ getState(): SessionState { diff --git a/packages/browser-pool/src/browser-pool.ts b/packages/browser-pool/src/browser-pool.ts index bddb11000b0b..1c2445004d27 100644 --- a/packages/browser-pool/src/browser-pool.ts +++ b/packages/browser-pool/src/browser-pool.ts @@ -355,8 +355,7 @@ export class BrowserPool< readonly #postPageCloseHooks: PostPageCloseHook[]; #pageCounter = 0; - // kept as TS-private rather than `#`: the page-close tests observe page tracking and - // controller retirement directly. Excluded from the public surface map either way. + // TS-private rather than `#`: tests observe page tracking and controller retirement directly private pages = new Map(); #pageIds = new WeakMap(); #startingBrowserControllers = new Set(); diff --git a/packages/core/src/proxy_configuration.ts b/packages/core/src/proxy_configuration.ts index 9e70124fb231..fe17dd15fcbb 100644 --- a/packages/core/src/proxy_configuration.ts +++ b/packages/core/src/proxy_configuration.ts @@ -39,6 +39,7 @@ export interface ProxyConfigurationOptions { */ validateRequired?: boolean; } + /** * Minimal contract that any object passed to a crawler as its `proxyConfiguration` * option must satisfy. diff --git a/packages/utils/src/internal.ts b/packages/utils/src/internal.ts index 4582eb7296dd..c0370cf94d80 100644 --- a/packages/utils/src/internal.ts +++ b/packages/utils/src/internal.ts @@ -7,4 +7,4 @@ export * from './internals/iterables.js'; export * from './internals/url.js'; export * from './internals/validation.js'; export * as schemas from './internals/schemas.js'; -export { parseSitemap } from './internals/sitemap.js'; +export { parseSitemap, type SitemapUrl } from './internals/sitemap.js'; diff --git a/website/sidebars.js b/website/sidebars.js index 03472e4acfe4..94abca3881e0 100644 --- a/website/sidebars.js +++ b/website/sidebars.js @@ -48,6 +48,7 @@ module.exports = { 'guides/impit-http-client/impit-http-client', 'guides/got-scraping', 'guides/typescript-project', + 'guides/public-api', 'guides/docker-images', 'guides/stagehand-crawler-guide', 'guides/running-in-web-server/running-in-web-server', From 88a1290efdefd760009232a2833f6b232fe717e9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Martin=20Ad=C3=A1mek?= Date: Tue, 29 Sep 2026 11:58:55 +0200 Subject: [PATCH 15/16] docs: drop v4-only changes from the v3 upgrading guide --- docs/upgrading/upgrading_v4.md | 71 +++++----------------------------- 1 file changed, 10 insertions(+), 61 deletions(-) diff --git a/docs/upgrading/upgrading_v4.md b/docs/upgrading/upgrading_v4.md index 2d213f05b3ac..9d7fd7f55365 100644 --- a/docs/upgrading/upgrading_v4.md +++ b/docs/upgrading/upgrading_v4.md @@ -952,7 +952,7 @@ This is intentional: these were never a supported API. If you relied on overridi The change spans, among others: -- **`BasicCrawler`** — `running`, `hasFinishedBefore`, `basicContextPipeline`, `unexpectedStop`, `requestHandlerTimeoutMillis`, `sameDomainDelayMillis`, `domainAccessedTime`, `handledRequestsCount`, `statusMessageLoggingInterval`, `statusMessageCallback`, `ignoreHttpErrorStatusCodes`, `taskLoopOptions` (was `autoscaledPoolOptions`), `autoscaledPool`, `respectRobotsTxtFile`, and the helpers `buildBasicContextPipeline`, `validateRequestUserData`, `pauseOnMigration`, `fetchNextRequest`, `delayRequest`, `handleRequest`, `timeoutAndRetry`, `isTaskReadyFunction`, `defaultIsFinishedFunction`, `requestFunctionErrorHandler`, `handleFailedRequestHandler`, `canRequestBeRetried` +- **`BasicCrawler`** — `running`, `hasFinishedBefore`, `unexpectedStop`, `requestHandlerTimeoutMillis`, `sameDomainDelayMillis`, `domainAccessedTime`, `handledRequestsCount`, `statusMessageLoggingInterval`, `statusMessageCallback`, `ignoreHttpErrorStatusCodes`, `taskLoopOptions` (was `autoscaledPoolOptions`), `autoscaledPool`, `respectRobotsTxtFile`, and the helpers `buildBasicContextPipeline`, `validateRequestUserData`, `pauseOnMigration`, `fetchNextRequest`, `delayRequest`, `handleRequest`, `timeoutAndRetry`, `isTaskReadyFunction`, `defaultIsFinishedFunction`, `requestFunctionErrorHandler`, `handleFailedRequestHandler`, `canRequestBeRetried` - **`HttpCrawler`** — `preNavigationHooks`, `postNavigationHooks`, `saveResponseCookies`, `navigationTimeoutMillis`, `suggestResponseEncoding`, `forceResponseEncoding`, `supportedMimeTypes`, and the helpers `requestFunction`, `parseResponse`, `getRequestOptions`, `encodeResponse`, `extendSupportedMimeTypes`, `handleRequestTimeout` - **`AutoscaledPool`** — the whole class is `@internal` in v4, so its members are not enumerated here; see [`AutoscaledPool` is no longer public API](#autoscaledpool-is-no-longer-public-api) - **`SessionPool`** — all pool internals (`log`, `maxPoolSize`, `createSessionFunction`, `keyValueStore`, `sessions`, `sessionMap`, `sessionOptions`, `persistStateKey`, `persistStateKeyValueStoreId`, `events`, `persistenceOptions`, `sessionReuseStrategy`, and the helpers `ensureInitialized`, `maybeLoadSessionPool`, `registerSession`, `createSession`, `hasSpaceForSession`, `pickSession`, `removeRetiredSessions`, `getRandomIndex`, `defaultCreateSessionFunction`) @@ -976,17 +976,13 @@ The `running` and `hasFinishedBefore` flags on `BasicCrawler` were internal run- ### Declarations that are now `@internal` -These are still present at runtime, but they are excluded from the documented surface and can change without a major version bump: `Session.getState()`, `SessionOptions.log`, `Router.getTimeoutSecs()`, `Router.getMaxTimeoutSecs()`, `Request.skippedReason`, `RestrictedCrawlingContext.id`, `ThrottlingRequestManager.drop()` and `ThrottlingRequestManager.innerManager`. +These are still present at runtime, but they are excluded from the documented surface and can change without a major version bump: `Session.getState()`, `SessionOptions.log`, `Request.skippedReason` and `RestrictedCrawlingContext.id`. `BLOCKED_STATUS_CODES` is now typed `readonly number[]`. Copy it (`[...BLOCKED_STATUS_CODES]`) if you were mutating it. -### `HttpCrawler`, `FileDownload` and `JSDOMCrawler` internals are no longer accessible to subclasses +### `HttpCrawler.isRequestBlocked` is now private -`HttpCrawler.isRequestBlocked` and `FileDownload.buildContextPipeline` are now `private`. Block detection is configured through `retryOnBlocked` and `blockedStatusCodes`; to add your own checks, throw a `SessionError` from a `postNavigationHook`. - -`HttpCrawler.buildContextPipeline` and `BrowserCrawler.buildContextPipeline` stay `protected` and overridable, and are supported extension points: they are the two levels that genuinely compose context stages, so override one of them to add your own. `BasicCrawler.buildContextPipeline` is gone — it only ever returned an empty pipeline — and the builders on `CheerioCrawler`, `JSDOMCrawler`, `LinkeDOMCrawler`, `PlaywrightCrawler`, `AdaptivePlaywrightCrawler`, `PuppeteerCrawler` and `StagehandCrawler` are now `#private`. If you were overriding one of those to add context members, use the `extendContext` option instead. It runs before navigation, so it cannot see navigation-dependent members such as `page` or `$`; read those in a `postNavigationHook` or the `requestHandler`. - -`HttpCrawler.getNavigationTimeoutMillis` and `HttpCrawler.createDefaultConcurrencySystem` stay `protected` and overridable, but are marked `@internal` — they are implementation seams shared with `@crawlee/cheerio`, `@crawlee/jsdom` and `@crawlee/linkedom`, and their signatures may change in a minor release. +Block detection is configured through `retryOnBlocked` and `blockedStatusCodes`; to add your own checks, throw a `SessionError` from a `postNavigationHook`. ### The protected `getMessageFromError()` returns `string` @@ -1385,8 +1381,6 @@ postNavigationHooks: [ If you called the standalone `playwrightUtils.handleCloudflareChallenge(page, url, session, options)` directly, note that the `session` parameter is gone - the v4 signature is `handleCloudflareChallenge(page, url, options)`, so an options object passed in the old fourth position would be silently ignored. -The `playwrightUtils` namespace also used to contain a nested, self-referential `playwrightUtils` object; it is gone. `handleCloudflareChallenge` was previously only reachable as `playwrightUtils.playwrightUtils.handleCloudflareChallenge` — `playwrightUtils.handleCloudflareChallenge(page, url, options)` now works as documented. - ### `PlaywrightLauncher` is no longer exported `PlaywrightLauncher` was an implementation detail of `launchPlaywright()` and the Playwright browser pools, and it is no longer part of `@crawlee/playwright`'s (or `crawlee`'s) public exports. Use `launchPlaywright(launchContext, configuration)` to get a `Browser`, or `playwrightBrowserPool()` / `remotePlaywrightBrowserPool()` when you need a pool. `PlaywrightLaunchContext` is still exported, so the options object can still be typed. @@ -1406,10 +1400,6 @@ The abstract `BrowserCrawler` base class also lost its third type parameter, `La `PlaywrightCrawler` no longer accepts a top-level `launcher` option, and `PlaywrightLaunchContext` no longer accepts `launchContextOptions`; both were silently ignored and are now reported as unknown options by the constructors' validation. Pass the browser type as `launchContext.launcher`, and persistent-context settings inside `launchContext.launchOptions`. -### `LauncherBrowserPoolOptions` and `LauncherRemoteBrowserPoolOptions` are no longer exported - -These two type aliases described the options accepted by `BrowserLauncher.createBrowserPool()` and `.createRemoteBrowserPool()`, both internal APIs. Use the per-library option types instead — `PlaywrightBrowserPoolOptions` / `RemotePlaywrightBrowserPoolOptions`, `PuppeteerBrowserPoolOptions` / `RemotePuppeteerBrowserPoolOptions`, `StagehandBrowserPoolOptions` / `RemoteStagehandBrowserPoolOptions` — which are the caller-facing types for `playwrightBrowserPool()`, `puppeteerBrowserPool()` and `stagehandBrowserPool()`. - ### Dead v3 fingerprinting types are removed `BrowserSpecification`, `GetFingerprintReturn` and `@crawlee/browser-pool`'s own `FingerprintGenerator` interface (which shadowed `fingerprint-generator`'s class of the same name) are no longer exported. They had no consumers — `BrowserPool.fingerprintGenerator` is typed by `fingerprint-generator`'s `FingerprintGenerator`, and `fingerprintGeneratorOptions` is still typed by `FingerprintGeneratorOptions`. @@ -1432,7 +1422,7 @@ The six hook arrays (`preLaunchHooks`, `postLaunchHooks`, `prePageCreateHooks`, `BrowserController.log` and `BrowserPlugin.log` are no longer part of the public type surface either. Subclasses inside `@crawlee/browser-pool` still use them, but they are not covered by backwards-compatibility guarantees — get your own logger from `serviceLocator.getLogger()`. -Several more members are `@internal`, and remain present at runtime only: `BrowserPool.fingerprintInjector`, `.fingerprintCache`, `.fingerprintOptions`, `.maxOpenBrowsers`, `.hasFreeBrowserSlot()`, `.hasActiveBrowserWithFreeCapacity()`; `BrowserController.isActive`, `.totalPages`, `.lastPageOpenedAt`, `.normalizeProxyOptions()`; the `PlaywrightPlugin` / `PuppeteerPlugin` `useRemoteConnection()` overrides; `RemoteBrowserPool.browserPool`; `RemoteBrowserPoolOptions.slotPollIntervalMillis`; `LaunchContextOptions.isRemote`; `AnonymizeProxySugarOptions`; and the `PlaywrightBrowser` constructor. The `_close` / `_kill` / `_newPage` / `_getCookies` / `_setCookies` hooks on `BrowserController` and the `_launch` / `addProxyToLaunchOptions` / `isChromiumBasedBrowser` hooks on `BrowserPlugin` are *not* in that list: they carry no release tag, on the abstract declarations and on the `PlaywrightController` / `PuppeteerController` / `PlaywrightPlugin` / `PuppeteerPlugin` overrides alike, and remain the extension contract for your own subclasses. +Several more members are `@internal`, and remain present at runtime only: `BrowserPool.fingerprintInjector`, `.fingerprintCache`, `.fingerprintOptions`; `BrowserController.isActive`, `.totalPages`, `.lastPageOpenedAt`, `.normalizeProxyOptions()`; `AnonymizeProxySugarOptions`; and the `PlaywrightBrowser` constructor. The `_close` / `_kill` / `_newPage` / `_getCookies` / `_setCookies` hooks on `BrowserController` and the `_launch` / `addProxyToLaunchOptions` / `isChromiumBasedBrowser` hooks on `BrowserPlugin` are *not* in that list: they carry no release tag, on the abstract declarations and on the `PlaywrightController` / `PuppeteerController` / `PlaywrightPlugin` / `PuppeteerPlugin` overrides alike, and remain the extension contract for your own subclasses. ## Only if you customize crawler statistics @@ -1520,21 +1510,9 @@ class MyClient extends BaseHttpClient { } ``` -#### `IResponseWithUrl` is removed and `ResponseWithUrl.url` is `readonly` - -The `IResponseWithUrl` interface exported by `@crawlee/http-client` has been removed. It only ever described `Response & { url: string }`, which is exactly what the concrete `ResponseWithUrl` class provides — use `ResponseWithUrl` (or plain `Response`) in its place. - -`ResponseWithUrl.url` is now `readonly`. Pass the URL through the constructor (`new ResponseWithUrl(body, { url, status, headers })`) rather than assigning to it afterwards; nothing in Crawlee ever mutated it. - -#### `fetch` is protected on the built-in HTTP clients - -`BaseHttpClient` has always declared `protected abstract fetch(input, init?)`, but `FetchHttpClient`, `GotScrapingHttpClient` and `ImpitHttpClient` accidentally re-declared it without the modifier, making the raw network call publicly reachable. All three are now `protected override`, matching the base contract. - -If you were calling `httpClient.fetch(request, options)` directly to bypass Crawlee's cookie and redirect handling, use `httpClient.sendRequest(request, options)` instead — it accepts `session`, `cookieJar`, `proxyUrl`, `timeoutMillis`, `signal` and `ignoreTlsErrors` and returns the final `Response`. Custom clients extending `BaseHttpClient` are unaffected: overriding `fetch` as `protected` was already the documented shape. - #### Removed `@crawlee/types` HTTP types -The `StreamOptions` and `RedirectHandler` types have been removed. They were the options and redirect-callback types for `BaseHttpClient.stream()`, which no longer exists in v4 — a custom HTTP client now only implements `sendRequest(request: Request, options?: SendRequestOptions)`. If you referenced either type, delete the reference; there is no replacement. +The `RedirectHandler` type has been removed. It was the redirect-callback type for `BaseHttpClient.stream()`, which no longer exists in v4 — a custom HTTP client now only implements `sendRequest(request: Request, options?: SendRequestOptions)`. If you referenced it, delete the reference; there is no replacement. The `BrowserLikeResponse` interface has been removed. It was a v3-era shim for reading `url()` and `headers()` off a got-style response during cookie handling, and has had no consumer since HTTP responses became standard `Response` objects. Read `response.url` and `response.headers` directly instead. @@ -1808,10 +1786,9 @@ await enqueueLinks({ urls, requestQueue }); await enqueueLinks({ urls, requestManager }); ``` -#### Removed loader and manager type aliases +#### Removed `UrlList` type alias -- `UrlList` and `NewUrlOptions` are gone; the signatures that used them now spell their types out inline (`(string | null)[]` and `{ request?: Request }` respectively). No behavioral change — replace the alias with the expansion if you referenced it. -- `RequestManagerOpener` is no longer exported, along with the `ThrottlingRequestManagerOptions.requestManagerOpener` option that took one. +`UrlList` is gone; the signatures that used it now spell the type out inline (`(string | null)[]`). No behavioral change — replace the alias with the expansion if you referenced it. ## Only if you configure or implement storage backends @@ -2029,21 +2006,6 @@ Because the in-memory queue lives entirely within a single process and is never `MemoryStorageBackend` never accepted `writeMetadata` (it has no on-disk format to begin with), so there is nothing to change there. -#### `FileSystemStorageBackend` exposes no directory fields - -`FileSystemStorageBackend` no longer exposes `localDataDirectory`, `datasetsDirectory`, `keyValueStoresDirectory` or `requestQueuesDirectory` as readable properties — the `StorageBackend` interface declares only methods, and these were never part of it. The on-disk layout is unchanged, so join the paths yourself from the directory you configured: - -```typescript -import { resolve } from 'node:path'; - -const localDataDirectory = './storage'; -const storageBackend = new FileSystemStorageBackend({ localDataDirectory }); - -const datasetsDirectory = resolve(localDataDirectory, 'datasets'); -const keyValueStoresDirectory = resolve(localDataDirectory, 'key_value_stores'); -const requestQueuesDirectory = resolve(localDataDirectory, 'request_queues'); -``` - #### Out-of-band key-value files (e.g. a hand-placed `INPUT.json`) Keys are literal. `aaa` and `aaa.json` are two distinct keys, and `FileSystemStorageBackend` never infers a key from a file's extension — in v3 a hand-placed `aaa.json` was readable as `aaa`, in v4 it is not. @@ -2063,12 +2025,7 @@ Beyond the literal keys, three v3 behaviors are gone: ### Storage internals are no longer part of the public API -Several declarations that were only ever implementation details of `Dataset`, `KeyValueStore`, `RequestQueue` and `StorageTransaction` are no longer exported or no longer documented: - -- `StorageStatsTracker` is no longer exported. Read the counters through the `stats` getter on each storage instead — `dataset.stats`, `store.stats`, `queue.stats` — whose types (`DatasetStats`, `KeyValueStoreStats`, `RequestQueueStats`) remain public. -- `resolveStorageIdentifier()` is no longer exported. Use `Dataset.open()` / `KeyValueStore.open()` / `RequestQueue.open()`, which accept the same `id` / `name` / `alias` identifier forms. -- `DatasetOptions`, `KeyValueStoreOptions` and `RequestQueueOptions` are internal. They only described the arguments of the storage constructors, which were already internal — always open storages through the static `open()` methods. -- `StorageTransaction.journal` and `StorageTransaction.policy`, along with the journal entry types (`JournalEntry`, `DatasetJournalEntry`, `KeyValueStoreJournalEntry`, `RequestQueueJournalEntry`, `JournaledRequest`), are internal. For read-only introspection of a transaction use the `StorageTransactionView` accessors: `datasetItems`, `enqueuedUrls`, `keyValueStoreChanges`. +`DatasetOptions`, `KeyValueStoreOptions` and `RequestQueueOptions` are internal. They only described the arguments of the storage constructors, which were already internal — always open storages through the static `open()` methods. ### `Dataset` field visibility now matches its siblings @@ -2076,10 +2033,6 @@ Several declarations that were only ever implementation details of `Dataset`, `K - `Dataset.id` and `Dataset.name` are `readonly`, matching `KeyValueStore` and `RequestQueue`. - `Dataset.log` has been removed. It was never read by Crawlee and `KeyValueStore` never had it; use your own logger, or `crawler.log` inside a request handler. -### `MemoryStorageBackend.createRequestQueueBackend()` returns the `RequestQueueBackend` interface - -It is now typed with the `RequestQueueBackend` interface from `@crawlee/types`, like its two sibling factories and like `FileSystemStorageBackend`. The runtime object is unchanged, but memory-only members (`listItems()`, `cacheKey`, `handledRequestCount`, `pendingRequestCount`, ...) are no longer visible through the return type. Read queue counts from `await getMetadata()`. - ## Only if you tuned autoscaling The `minConcurrency` / `maxConcurrency` / `maxRequestsPerMinute` crawler options work as before. This section matters when you used `autoscaledPoolOptions`, drove an `AutoscaledPool` directly, or configured snapshotting and system status. @@ -2156,8 +2109,6 @@ const crawler = new CheerioCrawler({ }); ``` -`ConcurrencySystem.desiredConcurrency` is a **read-only getter** — the setter is gone. The value is owned by the autoscaler, which recomputes it from the load signals on every tick, so any write was overwritten within one `autoscaleIntervalSecs`. Set the starting point with the `desiredConcurrency` constructor option, and retune a running system through `minConcurrency` / `maxConcurrency`, which both clamp `desiredConcurrency` into the new bounds immediately. - `crawler.pause()` resolves once the requests already in flight have settled, and leaves `run()` pending until you `resume()` — unlike `crawler.stop()`, which ends the run gracefully. One behavioral consequence of the split: pausing no longer suspends autoscaling, because the autoscaling interval belongs to the `ConcurrencySystem`, which knows nothing about its borrowers' pause state — deliberately, since other crawlers sharing it may still need scaling. A paused crawler's system keeps evaluating (and possibly scaling down) the desired concurrency and keeps emitting its periodic state log. Scaling *up* stays effectively blocked, as the current concurrency drains below the ratio required for a scale-up. To silence the system during a long pause, `stop()` it (if you own it) and `start()` it again before resuming; a restart discards the snapshots taken before it, so the pause is not mistaken for load. ##### If you were driving an `AutoscaledPool` directly @@ -2388,7 +2339,7 @@ The crawler-only parts of `@crawlee/core` moved to `@crawlee/basic`, so that `@c - `SessionPool`, `Session` and the session-pool constants - `Router` (with `RouterHandler`, `RouterRoutes` and `defaultRoute`) - the cookie helpers (`mergeCookies`, `getCookiesFromResponse`, …) and `parseRetryAfterHeader` -- `SitemapRequestLoader` (with `SitemapRequestLoaderOptions`) and `ThrottlingRequestManager` (with `ThrottlingRequestManagerOptions` and `RequestManagerOpener`) +- `SitemapRequestLoader` (with `SitemapRequestLoaderOptions`) and `ThrottlingRequestManager` (with `ThrottlingRequestManagerOptions`) - the `enqueueLinks()` option types (`EnqueueLinksOptions`, `ExtractLinksOptions`, `EnqueueUrlsOptions`, `RequestTransform`, `SkippedRequestCallback`) and the URL pattern types and helpers (`GlobInput`, `RegExpInput`, `UrlPatternInput`, `UrlPatternObject`, `constructUrlPatternObjects`, …) - the crawler-only error classes: `RetryRequestError`, `RequestThrottledError`, `PersistentRateLimitError`, `NavigationSkippedError`, `MissingSessionError`, `MissingRouteError`, `RequestHandlerError` and the `ContextPipeline*Error` types @@ -2489,7 +2440,6 @@ The full list of removed exports and members, for ctrl-F purposes. Where a repla - `PlainResponse` type (from `@crawlee/http`) — it wrapped the `got-scraping` response and is gone along with the rest of the old HTTP response surface (see [`CrawlingContext.response` is now of type `Response`](#crawlingcontextresponse-is-now-of-type-response)) - `checkStorageAccess`, `withCheckedStorageAccess` and the `RequestHandlerResult` type — superseded by the storage transaction mechanism; use `withDirectStorageAccess()` and `StorageTransactionView` (see [Storage writes in request handlers are transactional](#storage-writes-in-request-handlers-are-transactional)) - `CreateContextOptions` type (from `@crawlee/basic`) — a leftover of the pre-`ContextPipeline` context-creation design, unused by the library itself; context construction is now driven by `ContextPipeline` -- `BasicCrawler.basicContextPipeline` (public getter) — the basic half of the pipeline is an implementation detail of `BasicCrawler.run()`; compose behavior via `contextPipelineBuilder` / `ContextPipeline` composition instead - `ResponseLike` interface (from `@crawlee/core`) — a vestige of the pre-`fetch` HTTP implementation with no consumers; `getCookiesFromResponse()` has always taken a native `Response` - `UrlPatternObject` (from `@crawlee/core`) — the *compiled* form of a URL pattern, produced internally by the `enqueueLinks()` machinery. Keep using `UrlPatternInput` / `GlobInput` / `RegExpInput`, which are unchanged, and let the return type of the pattern helpers be inferred - `PERSIST_STATE_KEY` (from `@crawlee/core`) — to change where a session pool persists its state, pass `persistStateKey` to `SessionPool` @@ -2497,7 +2447,6 @@ The full list of removed exports and members, for ctrl-F purposes. Where a repla - `WithRequired` type (from `@crawlee/core`) — a bare TypeScript utility that was never crawlee vocabulary; `LoadedRequest` no longer goes through it, so declare your own if you were using it - `ErrorSnapshotter` and its `SnapshotResult` return type (from `@crawlee/core`) — an implementation detail of `ErrorTracker`. Error snapshotting is opt-in through `new Statistics({ saveErrorSnapshots: true })` (or `new ErrorTracker({ saveErrorSnapshots: true })`) - `ErrorTracker.errorSnapshotter` and `ErrorTracker.captureSnapshot()` — both private now. Snapshotting is driven from `addAsync()` on the first occurrence of each distinct error; the captured URLs surface as `firstErrorScreenshotUrl` / `firstErrorHtmlUrl` on the corresponding node of `errorTracker.result`, as before -- `assertBrowserPoolNotConfigured` (from `@crawlee/browser`) — an internal helper that produced the "cannot be combined with `browserPool`" error message; it moved to `@crawlee/utils/internal` - `MinimumSpeedStream` and `ByteCounterStream` (from `@crawlee/http`) — these `Transform` factories existed only to be piped inside `FileDownloadOptions.streamHandler`, which v4 removed. Compose your own `Transform` around `context.response.body` in the `requestHandler` instead; see the [file download with streams example](https://crawlee.dev/js/docs/examples/file-download-stream) - `HttpHook`, `FileDownloadHook`, `CheerioHook`, `JSDOMHook` and `LinkeDOMHook` types — see [Removed navigation hook type aliases](#removed-navigation-hook-type-aliases) - The `puppeteerClickElements` namespace (from `@crawlee/puppeteer`) — `clickElements`, `clickElementsAndInterceptNavigationRequests` and `isTargetRelevant` were internal helpers. Use `puppeteerUtils.enqueueLinksByClickingElements()`, or `context.enqueueLinksByClickingElements()` inside a request handler; the `EnqueueLinksByClickingElementsOptions` type is still exported directly from `@crawlee/puppeteer` From 2c26e6161306b6f71022b1c9df203ddc18fb2cd3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Martin=20Ad=C3=A1mek?= Date: Tue, 29 Sep 2026 15:14:31 +0200 Subject: [PATCH 16/16] refactor: run subclass buildContextPipeline overrides in browser crawlers, drop CheerioHook PlaywrightCrawler, PuppeteerCrawler and StagehandCrawler composed onto super.buildContextPipeline(), so an override of the protected BrowserCrawler builder in a subclass was silently skipped. Dispatch through this, like DOMCrawler does. Also remove the CheerioHook alias the upgrading guide already lists as removed, and restore the BrowserPool controller-set assertions via bracket access. --- docs/public-api/crawlee-cheerio.api.md | 5 ----- .../src/internals/cheerio-crawler.ts | 6 ------ .../src/internals/playwright-crawler.ts | 2 +- .../src/internals/puppeteer-crawler.ts | 2 +- .../src/internals/stagehand-crawler.ts | 2 +- test/browser-pool/browser-pool.test.ts | 3 +++ test/core/crawlers/playwright_crawler.test.ts | 20 +++++++++++++++++++ 7 files changed, 26 insertions(+), 14 deletions(-) diff --git a/docs/public-api/crawlee-cheerio.api.md b/docs/public-api/crawlee-cheerio.api.md index 01c91e6e92d1..ba45d343bcd7 100644 --- a/docs/public-api/crawlee-cheerio.api.md +++ b/docs/public-api/crawlee-cheerio.api.md @@ -12,7 +12,6 @@ import type { DOMCrawlingContext } from '@crawlee/http'; import type { ErrorHandler } from '@crawlee/http'; import type { GetUserDataFromRequest } from '@crawlee/http'; import type { HttpCrawlerOptions } from '@crawlee/http'; -import type { InternalHttpHook } from '@crawlee/http'; import type { RequestHandler } from '@crawlee/http'; import type { RouterHandler } from '@crawlee/http'; import type { RouterRoutes } from '@crawlee/http'; @@ -40,10 +39,6 @@ export type CheerioErrorHandler> = ErrorHandler & ContextExtension>; -// @public (undocumented) -export type CheerioHook = InternalHttpHook>; - // @public (undocumented) export interface CheerioParseResult { // (undocumented) diff --git a/packages/cheerio-crawler/src/internals/cheerio-crawler.ts b/packages/cheerio-crawler/src/internals/cheerio-crawler.ts index c5c64fb8dcca..291e89d1272f 100644 --- a/packages/cheerio-crawler/src/internals/cheerio-crawler.ts +++ b/packages/cheerio-crawler/src/internals/cheerio-crawler.ts @@ -4,7 +4,6 @@ import type { ErrorHandler, GetUserDataFromRequest, HttpCrawlerOptions, - InternalHttpHook, RequestHandler, RouterHandler, RouterRoutes, @@ -41,11 +40,6 @@ export interface CheerioCrawlerOptions< 'contextPipelineBuilder' > {} -export type CheerioHook< - UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler - JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler -> = InternalHttpHook>; - export interface CheerioCrawlingContext< UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler diff --git a/packages/playwright-crawler/src/internals/playwright-crawler.ts b/packages/playwright-crawler/src/internals/playwright-crawler.ts index 18e4b27ff17f..408568ffd799 100644 --- a/packages/playwright-crawler/src/internals/playwright-crawler.ts +++ b/packages/playwright-crawler/src/internals/playwright-crawler.ts @@ -258,7 +258,7 @@ export class PlaywrightCrawler< } #buildContextPipeline(): ContextPipeline { - return super.buildContextPipeline().compose(this.enhanceContext.bind(this)); + return this.buildContextPipeline().compose(this.enhanceContext.bind(this)); } protected override async navigationHandler( diff --git a/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts b/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts index 7e4cc141e6cc..e533c72140d3 100644 --- a/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts +++ b/packages/puppeteer-crawler/src/internals/puppeteer-crawler.ts @@ -255,7 +255,7 @@ export class PuppeteerCrawler< } #buildContextPipeline(): ContextPipeline { - return super.buildContextPipeline().compose(this.enhanceContext.bind(this)); + return this.buildContextPipeline().compose(this.enhanceContext.bind(this)); } private async enhanceContext(context: BrowserCrawlingContext) { diff --git a/packages/stagehand-crawler/src/internals/stagehand-crawler.ts b/packages/stagehand-crawler/src/internals/stagehand-crawler.ts index e5a8d8de614a..026b434b54e3 100644 --- a/packages/stagehand-crawler/src/internals/stagehand-crawler.ts +++ b/packages/stagehand-crawler/src/internals/stagehand-crawler.ts @@ -469,7 +469,7 @@ export class StagehandCrawler< } #buildContextPipeline(): ContextPipeline { - return super.buildContextPipeline().compose(this.setUpStagehand.bind(this)); + return this.buildContextPipeline().compose(this.setUpStagehand.bind(this)); } /** diff --git a/test/browser-pool/browser-pool.test.ts b/test/browser-pool/browser-pool.test.ts index f9a0f8db9cdb..69fea27ec0eb 100644 --- a/test/browser-pool/browser-pool.test.ts +++ b/test/browser-pool/browser-pool.test.ts @@ -142,6 +142,8 @@ describe.each([ await browserPool.destroy(); expect(browserController.close).toHaveBeenCalled(); + expect(browserPool['activeBrowserControllers'].size).toBe(0); + expect(browserPool['retiredBrowserControllers'].size).toBe(0); expect(browserPool['browserKillerInterval']).toBeUndefined(); }); }); @@ -246,6 +248,7 @@ describe.each([ await browserPool.newPageInNewBrowser(); await browserPool.newPageInNewBrowser(); + expect(browserPool['activeBrowserControllers'].size).toBe(3); expect(plugin.launch).toHaveBeenCalledTimes(3); }); diff --git a/test/core/crawlers/playwright_crawler.test.ts b/test/core/crawlers/playwright_crawler.test.ts index 5cddb476eca5..5a45c0c32c17 100644 --- a/test/core/crawlers/playwright_crawler.test.ts +++ b/test/core/crawlers/playwright_crawler.test.ts @@ -323,6 +323,26 @@ describe('PlaywrightCrawler', () => { await playwrightCrawler.run(); }); + test('runs a buildContextPipeline override from a subclass', async () => { + class MyCrawler extends PlaywrightCrawler { + protected override buildContextPipeline() { + return super.buildContextPipeline().compose(async () => ({ myField: 123 })); + } + } + + let myField: unknown; + const crawler = new MyCrawler({ + requestList, + maxRequestRetries: 0, + requestHandler: async (context) => { + myField = (context as unknown as { myField: number }).myField; + }, + }); + await crawler.run(); + + expect(myField).toBe(123); + }); + test('validates userData against the router schema when adding requests', async () => { const router = createPlaywrightRouter({ DETAIL: z.object({ id: z.string() }),