refactor(browser fetchers): Make all the type hints dynamic + Faster validation

Made the code shorter by an additional ~200 lines and easier to maintain in return for making the arguments autocompletion bad for shells that don't check for dynamic type hints like IPython.
This commit is contained in:
Karim shoair
2025-11-25 20:18:36 +02:00
parent 9e85311ff6
commit f0b01fc253
7 changed files with 364 additions and 637 deletions
+1
View File
@@ -12,6 +12,7 @@ from typing import (
Callable, Callable,
Dict, Dict,
Generator, Generator,
Generic,
Iterable, Iterable,
List, List,
Set, Set,
+99 -94
View File
@@ -2,15 +2,15 @@ from time import time
from asyncio import sleep as asyncio_sleep, Lock from asyncio import sleep as asyncio_sleep, Lock
from camoufox import DefaultAddons from camoufox import DefaultAddons
from playwright.sync_api._generated import Page
from playwright.sync_api import ( from playwright.sync_api import (
Page,
Frame, Frame,
BrowserContext, BrowserContext,
Playwright, Playwright,
Response as SyncPlaywrightResponse, Response as SyncPlaywrightResponse,
) )
from playwright.async_api._generated import Page as AsyncPage
from playwright.async_api import ( from playwright.async_api import (
Page as AsyncPage,
Frame as AsyncFrame, Frame as AsyncFrame,
Playwright as AsyncPlaywright, Playwright as AsyncPlaywright,
Response as AsyncPlaywrightResponse, Response as AsyncPlaywrightResponse,
@@ -70,7 +70,7 @@ class SyncSession:
timeout: int | float, timeout: int | float,
extra_headers: Optional[Dict[str, str]], extra_headers: Optional[Dict[str, str]],
disable_resources: bool, disable_resources: bool,
) -> PageInfo: # pragma: no cover ) -> PageInfo[Page]: # pragma: no cover
"""Get a new page to use""" """Get a new page to use"""
# No need to check if a page is available or not in sync code because the code blocked before reaching here till the page closed, ofc. # No need to check if a page is available or not in sync code because the code blocked before reaching here till the page closed, ofc.
@@ -116,7 +116,7 @@ class SyncSession:
self._wait_for_networkidle(page) self._wait_for_networkidle(page)
@staticmethod @staticmethod
def _create_response_handler(page_info: PageInfo, response_container: List) -> Callable: def _create_response_handler(page_info: PageInfo[Page], response_container: List) -> Callable:
"""Create a response handler that captures the final navigation response. """Create a response handler that captures the final navigation response.
:param page_info: The PageInfo object containing the page :param page_info: The PageInfo object containing the page
@@ -175,7 +175,7 @@ class AsyncSession:
timeout: int | float, timeout: int | float,
extra_headers: Optional[Dict[str, str]], extra_headers: Optional[Dict[str, str]],
disable_resources: bool, disable_resources: bool,
) -> PageInfo: # pragma: no cover ) -> PageInfo[AsyncPage]: # pragma: no cover
"""Get a new page to use""" """Get a new page to use"""
if TYPE_CHECKING: if TYPE_CHECKING:
assert self.context is not None, "Browser context not initialized" assert self.context is not None, "Browser context not initialized"
@@ -232,7 +232,7 @@ class AsyncSession:
await self._wait_for_networkidle(page) await self._wait_for_networkidle(page)
@staticmethod @staticmethod
def _create_response_handler(page_info: PageInfo, response_container: List) -> Callable: def _create_response_handler(page_info: PageInfo[AsyncPage], response_container: List) -> Callable:
"""Create an async response handler that captures the final navigation response. """Create an async response handler that captures the final navigation response.
:param page_info: The PageInfo object containing the page :param page_info: The PageInfo object containing the page
@@ -253,130 +253,135 @@ class AsyncSession:
class DynamicSessionMixin: class DynamicSessionMixin:
def __validate__(self, **params): def __validate__(self, **params):
if "__max_pages" in params:
params["max_pages"] = params.pop("__max_pages")
config = validate(params, model=PlaywrightConfig) config = validate(params, model=PlaywrightConfig)
self.max_pages = config.max_pages self._max_pages = config.max_pages
self.headless = config.headless self._headless = config.headless
self.hide_canvas = config.hide_canvas self._hide_canvas = config.hide_canvas
self.disable_webgl = config.disable_webgl self._disable_webgl = config.disable_webgl
self.real_chrome = config.real_chrome self._real_chrome = config.real_chrome
self.stealth = config.stealth self._stealth = config.stealth
self.google_search = config.google_search self._google_search = config.google_search
self.wait = config.wait self._wait = config.wait
self.proxy = config.proxy self._proxy = config.proxy
self.locale = config.locale self._locale = config.locale
self.extra_headers = config.extra_headers self._extra_headers = config.extra_headers
self.useragent = config.useragent self._useragent = config.useragent
self.timeout = config.timeout self._timeout = config.timeout
self.cookies = config.cookies self._cookies = config.cookies
self.disable_resources = config.disable_resources self._disable_resources = config.disable_resources
self.cdp_url = config.cdp_url self._cdp_url = config.cdp_url
self.network_idle = config.network_idle self._network_idle = config.network_idle
self.load_dom = config.load_dom self._load_dom = config.load_dom
self.wait_selector = config.wait_selector self._wait_selector = config.wait_selector
self.init_script = config.init_script self._init_script = config.init_script
self.wait_selector_state = config.wait_selector_state self._wait_selector_state = config.wait_selector_state
self.extra_flags = config.extra_flags self._extra_flags = config.extra_flags
self.selector_config = config.selector_config self._selector_config = config.selector_config
self.additional_args = config.additional_args self._additional_args = config.additional_args
self.page_action = config.page_action self._page_action = config.page_action
self.user_data_dir = config.user_data_dir self._user_data_dir = config.user_data_dir
self._headers_keys = {header.lower() for header in self.extra_headers.keys()} if self.extra_headers else set() self._headers_keys = {header.lower() for header in self._extra_headers.keys()} if self._extra_headers else set()
self.__initiate_browser_options__() self.__initiate_browser_options__()
def __initiate_browser_options__(self): def __initiate_browser_options__(self):
if TYPE_CHECKING: if TYPE_CHECKING:
assert isinstance(self.proxy, tuple) assert isinstance(self._proxy, tuple)
if not self.cdp_url: if not self._cdp_url:
# `launch_options` is used with persistent context # `launch_options` is used with persistent context
self.launch_options = dict( self.launch_options = dict(
_launch_kwargs( _launch_kwargs(
self.headless, self._headless,
self.proxy, self._proxy,
self.locale, self._locale,
tuple(self.extra_headers.items()) if self.extra_headers else tuple(), tuple(self._extra_headers.items()) if self._extra_headers else tuple(),
self.useragent, self._useragent,
self.real_chrome, self._real_chrome,
self.stealth, self._stealth,
self.hide_canvas, self._hide_canvas,
self.disable_webgl, self._disable_webgl,
tuple(self.extra_flags) if self.extra_flags else tuple(), tuple(self._extra_flags) if self._extra_flags else tuple(),
) )
) )
self.launch_options["extra_http_headers"] = dict(self.launch_options["extra_http_headers"]) self.launch_options["extra_http_headers"] = dict(self.launch_options["extra_http_headers"])
self.launch_options["proxy"] = dict(self.launch_options["proxy"]) or None self.launch_options["proxy"] = dict(self.launch_options["proxy"]) or None
self.launch_options["user_data_dir"] = self.user_data_dir self.launch_options["user_data_dir"] = self._user_data_dir
self.launch_options.update(cast(Dict, self.additional_args)) self.launch_options.update(cast(Dict, self._additional_args))
self.context_options = dict() self.context_options = dict()
else: else:
# while `context_options` is left to be used when cdp mode is enabled # while `context_options` is left to be used when cdp mode is enabled
self.launch_options = dict() self.launch_options = dict()
self.context_options = dict( self.context_options = dict(
_context_kwargs( _context_kwargs(
self.proxy, self._proxy,
self.locale, self._locale,
tuple(self.extra_headers.items()) if self.extra_headers else tuple(), tuple(self._extra_headers.items()) if self._extra_headers else tuple(),
self.useragent, self._useragent,
self.stealth, self._stealth,
) )
) )
self.context_options["extra_http_headers"] = dict(self.context_options["extra_http_headers"]) self.context_options["extra_http_headers"] = dict(self.context_options["extra_http_headers"])
self.context_options["proxy"] = dict(self.context_options["proxy"]) or None self.context_options["proxy"] = dict(self.context_options["proxy"]) or None
self.context_options.update(cast(Dict, self.additional_args)) self.context_options.update(cast(Dict, self._additional_args))
class StealthySessionMixin: class StealthySessionMixin:
def __validate__(self, **params): def __validate__(self, **params):
if "__max_pages" in params:
params["max_pages"] = params.pop("__max_pages")
config: CamoufoxConfig = validate(params, model=CamoufoxConfig) config: CamoufoxConfig = validate(params, model=CamoufoxConfig)
self.max_pages = config.max_pages self._max_pages = config.max_pages
self.headless = config.headless self._headless = config.headless
self.block_images = config.block_images self._block_images = config.block_images
self.disable_resources = config.disable_resources self._disable_resources = config.disable_resources
self.block_webrtc = config.block_webrtc self._block_webrtc = config.block_webrtc
self.allow_webgl = config.allow_webgl self._allow_webgl = config.allow_webgl
self.network_idle = config.network_idle self._network_idle = config.network_idle
self.load_dom = config.load_dom self._load_dom = config.load_dom
self.humanize = config.humanize self._humanize = config.humanize
self.solve_cloudflare = config.solve_cloudflare self._solve_cloudflare = config.solve_cloudflare
self.wait = config.wait self._wait = config.wait
self.timeout = config.timeout self._timeout = config.timeout
self.page_action = config.page_action self._page_action = config.page_action
self.wait_selector = config.wait_selector self._wait_selector = config.wait_selector
self.init_script = config.init_script self._init_script = config.init_script
self.addons = config.addons self._addons = config.addons
self.wait_selector_state = config.wait_selector_state self._wait_selector_state = config.wait_selector_state
self.cookies = config.cookies self._cookies = config.cookies
self.google_search = config.google_search self._google_search = config.google_search
self.extra_headers = config.extra_headers self._extra_headers = config.extra_headers
self.proxy = config.proxy self._proxy = config.proxy
self.os_randomize = config.os_randomize self._os_randomize = config.os_randomize
self.disable_ads = config.disable_ads self._disable_ads = config.disable_ads
self.geoip = config.geoip self._geoip = config.geoip
self.selector_config = config.selector_config self._selector_config = config.selector_config
self.additional_args = config.additional_args self._additional_args = config.additional_args
self.page_action = config.page_action self._user_data_dir = config.user_data_dir
self.user_data_dir = config.user_data_dir self._headers_keys = {header.lower() for header in self._extra_headers.keys()} if self._extra_headers else set()
self._headers_keys = {header.lower() for header in self.extra_headers.keys()} if self.extra_headers else set()
self.__initiate_browser_options__() self.__initiate_browser_options__()
def __initiate_browser_options__(self): def __initiate_browser_options__(self):
"""Initiate browser options.""" """Initiate browser options."""
self.launch_options: Dict[str, Any] = generate_launch_options( self.launch_options: Dict[str, Any] = generate_launch_options(
**{ **{
"geoip": self.geoip, "geoip": self._geoip,
"proxy": dict(self.proxy) if self.proxy and isinstance(self.proxy, tuple) else self.proxy, "proxy": dict(self._proxy) if self._proxy and isinstance(self._proxy, tuple) else self._proxy,
"addons": self.addons, "addons": self._addons,
"exclude_addons": [] if self.disable_ads else [DefaultAddons.UBO], "exclude_addons": [] if self._disable_ads else [DefaultAddons.UBO],
"headless": self.headless, "headless": self._headless,
"humanize": True if self.solve_cloudflare else self.humanize, "humanize": True if self._solve_cloudflare else self._humanize,
"i_know_what_im_doing": True, # To turn warnings off with the user configurations "i_know_what_im_doing": True, # To turn warnings off with the user configurations
"allow_webgl": self.allow_webgl, "allow_webgl": self._allow_webgl,
"block_webrtc": self.block_webrtc, "block_webrtc": self._block_webrtc,
"block_images": self.block_images, # Careful! it makes some websites don't finish loading at all like stackoverflow even in headful mode. "block_images": self._block_images, # Careful! it makes some websites don't finish loading at all like stackoverflow even in headful mode.
"os": None if self.os_randomize else get_os_name(), "os": None if self._os_randomize else get_os_name(),
"user_data_dir": self.user_data_dir, "user_data_dir": self._user_data_dir,
"ff_version": __ff_version_str__, "ff_version": __ff_version_str__,
"firefox_user_prefs": { "firefox_user_prefs": {
# This is what enabling `enable_cache` does internally, so we do it from here instead # This is what enabling `enable_cache` does internally, so we do it from here instead
@@ -386,7 +391,7 @@ class StealthySessionMixin:
"browser.cache.disk_cache_ssl": True, "browser.cache.disk_cache_ssl": True,
"browser.cache.disk.smart_size.enabled": True, "browser.cache.disk.smart_size.enabled": True,
}, },
**cast(Dict, self.additional_args), **cast(Dict, self._additional_args),
} }
) )
+80 -267
View File
@@ -14,58 +14,47 @@ from playwright.async_api import (
BrowserContext as AsyncBrowserContext, BrowserContext as AsyncBrowserContext,
) )
from ._validators import validate_fetch as _validate, CamoufoxConfig
from ._base import SyncSession, AsyncSession, StealthySessionMixin
from scrapling.core.utils import log from scrapling.core.utils import log
from scrapling.core._types import ( from ._types import CamoufoxSession, CamoufoxFetchParams
Any, from scrapling.core._types import Any, Unpack, TYPE_CHECKING
Dict, from ._base import SyncSession, AsyncSession, StealthySessionMixin
List, from ._validators import validate_fetch as _validate, CamoufoxConfig
Optional, from scrapling.engines.toolbelt.convertor import Response, ResponseFactory
Callable,
TYPE_CHECKING,
SelectorWaitStates,
)
from scrapling.engines.toolbelt.convertor import (
Response,
ResponseFactory,
)
from scrapling.engines.toolbelt.fingerprints import generate_convincing_referer from scrapling.engines.toolbelt.fingerprints import generate_convincing_referer
__CF_PATTERN__ = re_compile("challenges.cloudflare.com/cdn-cgi/challenge-platform/.*") __CF_PATTERN__ = re_compile("challenges.cloudflare.com/cdn-cgi/challenge-platform/.*")
_UNSET: Any = object()
class StealthySession(StealthySessionMixin, SyncSession): class StealthySession(StealthySessionMixin, SyncSession):
"""A Stealthy session manager with page pooling.""" """A Stealthy session manager with page pooling."""
__slots__ = ( __slots__ = (
"max_pages", "_max_pages",
"headless", "_headless",
"block_images", "_block_images",
"disable_resources", "_disable_resources",
"block_webrtc", "_block_webrtc",
"allow_webgl", "_allow_webgl",
"network_idle", "_network_idle",
"load_dom", "_load_dom",
"humanize", "_humanize",
"solve_cloudflare", "_solve_cloudflare",
"wait", "_wait",
"timeout", "_timeout",
"page_action", "_page_action",
"wait_selector", "_wait_selector",
"init_script", "_init_script",
"addons", "_addons",
"wait_selector_state", "_wait_selector_state",
"cookies", "_cookies",
"google_search", "_google_search",
"extra_headers", "_extra_headers",
"proxy", "_proxy",
"os_randomize", "_os_randomize",
"disable_ads", "_disable_ads",
"geoip", "_geoip",
"selector_config", "_selector_config",
"additional_args", "_additional_args",
"playwright", "playwright",
"browser", "browser",
"context", "context",
@@ -73,38 +62,10 @@ class StealthySession(StealthySessionMixin, SyncSession):
"_closed", "_closed",
"launch_options", "launch_options",
"_headers_keys", "_headers_keys",
"_user_data_dir",
) )
def __init__( def __init__(self, **kwargs: Unpack[CamoufoxSession]):
self,
__max_pages: int = 1,
headless: bool = True, # noqa: F821
block_images: bool = False,
disable_resources: bool = False,
block_webrtc: bool = False,
allow_webgl: bool = True,
network_idle: bool = False,
load_dom: bool = True,
humanize: bool | float = True,
solve_cloudflare: bool = False,
wait: int | float = 0,
timeout: int | float = 30000,
page_action: Optional[Callable] = None,
wait_selector: Optional[str] = None,
init_script: Optional[str] = None,
addons: Optional[List[str]] = None,
wait_selector_state: SelectorWaitStates = "attached",
cookies: Optional[List[Dict]] = None,
google_search: bool = True,
extra_headers: Optional[Dict[str, str]] = None,
proxy: Optional[str | Dict[str, str]] = None,
os_randomize: bool = False,
disable_ads: bool = False,
geoip: bool = False,
user_data_dir: str = "",
selector_config: Optional[Dict] = None,
additional_args: Optional[Dict] = None,
):
"""A Browser session manager with page pooling """A Browser session manager with page pooling
:param headless: Run the browser in headless/hidden (default), or headful/visible mode. :param headless: Run the browser in headless/hidden (default), or headful/visible mode.
@@ -138,50 +99,21 @@ class StealthySession(StealthySessionMixin, SyncSession):
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. :param selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param additional_args: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings. :param additional_args: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings.
""" """
self.__validate__(**kwargs)
self.__validate__( super().__init__(max_pages=self._max_pages)
wait=wait,
proxy=proxy,
geoip=geoip,
addons=addons,
timeout=timeout,
cookies=cookies,
headless=headless,
humanize=humanize,
load_dom=load_dom,
max_pages=__max_pages,
disable_ads=disable_ads,
allow_webgl=allow_webgl,
page_action=page_action,
init_script=init_script,
network_idle=network_idle,
block_images=block_images,
block_webrtc=block_webrtc,
os_randomize=os_randomize,
user_data_dir=user_data_dir,
wait_selector=wait_selector,
google_search=google_search,
extra_headers=extra_headers,
additional_args=additional_args,
selector_config=selector_config,
solve_cloudflare=solve_cloudflare,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
)
super().__init__(max_pages=self.max_pages)
def __create__(self): def __create__(self):
"""Create a browser for this instance and context.""" """Create a browser for this instance and context."""
self.playwright = sync_playwright().start() self.playwright = sync_playwright().start()
self.context = self.playwright.firefox.launch_persistent_context(**self.launch_options) self.context = self.playwright.firefox.launch_persistent_context(**self.launch_options)
if self.init_script: # pragma: no cover if self._init_script: # pragma: no cover
self.context.add_init_script(path=self.init_script) self.context.add_init_script(path=self._init_script)
if self.cookies: # pragma: no cover if self._cookies: # pragma: no cover
self.context.add_cookies(self.cookies) self.context.add_cookies(self._cookies)
def _solve_cloudflare(self, page: Page) -> None: # pragma: no cover def _cloudflare_solver(self, page: Page) -> None: # pragma: no cover
"""Solve the cloudflare challenge displayed on the playwright page passed """Solve the cloudflare challenge displayed on the playwright page passed
:param page: The targeted page :param page: The targeted page
@@ -247,59 +179,28 @@ class StealthySession(StealthySessionMixin, SyncSession):
log.info("Cloudflare captcha is solved") log.info("Cloudflare captcha is solved")
return return
def fetch( def fetch(self, url: str, **kwargs: Unpack[CamoufoxFetchParams]) -> Response:
self,
url: str,
google_search: bool = _UNSET,
timeout: int | float = _UNSET,
wait: int | float = _UNSET,
page_action: Optional[Callable] = _UNSET,
extra_headers: Optional[Dict[str, str]] = _UNSET,
disable_resources: bool = _UNSET,
wait_selector: Optional[str] = _UNSET,
wait_selector_state: SelectorWaitStates = _UNSET,
network_idle: bool = _UNSET,
load_dom: bool = _UNSET,
solve_cloudflare: bool = _UNSET,
selector_config: Optional[Dict] = _UNSET,
) -> Response:
"""Opens up the browser and do your request based on your chosen options. """Opens up the browser and do your request based on your chosen options.
:param url: The Target url. :param url: The Target url.
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. :param kwargs: Additional keyword arguments including:
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - page_action: Added for automation. A function that takes the `page` object and does the automation you need.
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
- disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites.
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
:param wait_selector: Wait for a specific CSS selector to be in a specific state. - wait_selector: Wait for a specific CSS selector to be in a specific state.
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. - network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute.
:param solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you. - solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you.
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. - selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:return: A `Response` object. :return: A `Response` object.
""" """
params = _validate( params = _validate(kwargs, self, CamoufoxConfig)
[
("google_search", google_search, self.google_search),
("timeout", timeout, self.timeout),
("wait", wait, self.wait),
("page_action", page_action, self.page_action),
("extra_headers", extra_headers, self.extra_headers),
("disable_resources", disable_resources, self.disable_resources),
("wait_selector", wait_selector, self.wait_selector),
("wait_selector_state", wait_selector_state, self.wait_selector_state),
("network_idle", network_idle, self.network_idle),
("load_dom", load_dom, self.load_dom),
("solve_cloudflare", solve_cloudflare, self.solve_cloudflare),
("selector_config", selector_config, self.selector_config),
],
CamoufoxConfig,
_UNSET,
)
if self._closed: # pragma: no cover if self._closed: # pragma: no cover
raise RuntimeError("Context manager has been closed") raise RuntimeError("Context manager has been closed")
@@ -322,7 +223,7 @@ class StealthySession(StealthySessionMixin, SyncSession):
raise RuntimeError(f"Failed to get response for {url}") raise RuntimeError(f"Failed to get response for {url}")
if params.solve_cloudflare: if params.solve_cloudflare:
self._solve_cloudflare(page_info.page) self._cloudflare_solver(page_info.page)
# Make sure the page is fully loaded after the captcha # Make sure the page is fully loaded after the captcha
self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle) self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
@@ -360,36 +261,7 @@ class StealthySession(StealthySessionMixin, SyncSession):
class AsyncStealthySession(StealthySessionMixin, AsyncSession): class AsyncStealthySession(StealthySessionMixin, AsyncSession):
"""A Stealthy session manager with page pooling.""" """A Stealthy session manager with page pooling."""
def __init__( def __init__(self, **kwargs: Unpack[CamoufoxSession]):
self,
max_pages: int = 1,
headless: bool = True, # noqa: F821
block_images: bool = False,
disable_resources: bool = False,
block_webrtc: bool = False,
allow_webgl: bool = True,
network_idle: bool = False,
load_dom: bool = True,
humanize: bool | float = True,
solve_cloudflare: bool = False,
wait: int | float = 0,
timeout: int | float = 30000,
page_action: Optional[Callable] = None,
wait_selector: Optional[str] = None,
init_script: Optional[str] = None,
addons: Optional[List[str]] = None,
wait_selector_state: SelectorWaitStates = "attached",
cookies: Optional[List[Dict]] = None,
google_search: bool = True,
extra_headers: Optional[Dict[str, str]] = None,
proxy: Optional[str | Dict[str, str]] = None,
os_randomize: bool = False,
disable_ads: bool = False,
geoip: bool = False,
user_data_dir: str = "",
selector_config: Optional[Dict] = None,
additional_args: Optional[Dict] = None,
):
"""A Browser session manager with page pooling """A Browser session manager with page pooling
:param headless: Run the browser in headless/hidden (default), or headful/visible mode. :param headless: Run the browser in headless/hidden (default), or headful/visible mode.
@@ -424,36 +296,8 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. :param selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param additional_args: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings. :param additional_args: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings.
""" """
self.__validate__( self.__validate__(**kwargs)
wait=wait, super().__init__(max_pages=self._max_pages)
proxy=proxy,
geoip=geoip,
addons=addons,
timeout=timeout,
cookies=cookies,
headless=headless,
load_dom=load_dom,
humanize=humanize,
max_pages=max_pages,
disable_ads=disable_ads,
allow_webgl=allow_webgl,
page_action=page_action,
init_script=init_script,
network_idle=network_idle,
block_images=block_images,
block_webrtc=block_webrtc,
os_randomize=os_randomize,
wait_selector=wait_selector,
google_search=google_search,
extra_headers=extra_headers,
user_data_dir=user_data_dir,
additional_args=additional_args,
selector_config=selector_config,
solve_cloudflare=solve_cloudflare,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
)
super().__init__(max_pages=self.max_pages)
async def __create__(self): async def __create__(self):
"""Create a browser for this instance and context.""" """Create a browser for this instance and context."""
@@ -462,13 +306,13 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
**self.launch_options **self.launch_options
) )
if self.init_script: # pragma: no cover if self._init_script: # pragma: no cover
await self.context.add_init_script(path=self.init_script) await self.context.add_init_script(path=self._init_script)
if self.cookies: if self._cookies:
await self.context.add_cookies(self.cookies) # pyright: ignore [reportArgumentType] await self.context.add_cookies(self._cookies) # pyright: ignore [reportArgumentType]
async def _solve_cloudflare(self, page: async_Page): # pragma: no cover async def _cloudflare_solver(self, page: async_Page): # pragma: no cover
"""Solve the cloudflare challenge displayed on the playwright page passed. The async version """Solve the cloudflare challenge displayed on the playwright page passed. The async version
:param page: The async targeted page :param page: The async targeted page
@@ -534,59 +378,28 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
log.info("Cloudflare captcha is solved") log.info("Cloudflare captcha is solved")
return return
async def fetch( async def fetch(self, url: str, **kwargs: Unpack[CamoufoxFetchParams]) -> Response:
self,
url: str,
google_search: bool = _UNSET,
timeout: int | float = _UNSET,
wait: int | float = _UNSET,
page_action: Optional[Callable] = _UNSET,
extra_headers: Optional[Dict[str, str]] = _UNSET,
disable_resources: bool = _UNSET,
wait_selector: Optional[str] = _UNSET,
wait_selector_state: SelectorWaitStates = _UNSET,
network_idle: bool = _UNSET,
load_dom: bool = _UNSET,
solve_cloudflare: bool = _UNSET,
selector_config: Optional[Dict] = _UNSET,
) -> Response:
"""Opens up the browser and do your request based on your chosen options. """Opens up the browser and do your request based on your chosen options.
:param url: The Target url. :param url: The Target url.
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. :param kwargs: Additional keyword arguments including:
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - page_action: Added for automation. A function that takes the `page` object and does the automation you need.
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
- disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites.
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
:param wait_selector: Wait for a specific CSS selector to be in a specific state. - wait_selector: Wait for a specific CSS selector to be in a specific state.
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. - network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute.
:param solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you. - solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you.
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. - selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:return: A `Response` object. :return: A `Response` object.
""" """
params = _validate( params = _validate(kwargs, self, CamoufoxConfig)
[
("google_search", google_search, self.google_search),
("timeout", timeout, self.timeout),
("wait", wait, self.wait),
("page_action", page_action, self.page_action),
("extra_headers", extra_headers, self.extra_headers),
("disable_resources", disable_resources, self.disable_resources),
("wait_selector", wait_selector, self.wait_selector),
("wait_selector_state", wait_selector_state, self.wait_selector_state),
("network_idle", network_idle, self.network_idle),
("load_dom", load_dom, self.load_dom),
("solve_cloudflare", solve_cloudflare, self.solve_cloudflare),
("selector_config", selector_config, self.selector_config),
],
CamoufoxConfig,
_UNSET,
)
if self._closed: # pragma: no cover if self._closed: # pragma: no cover
raise RuntimeError("Context manager has been closed") raise RuntimeError("Context manager has been closed")
@@ -613,7 +426,7 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
raise RuntimeError(f"Failed to get response for {url}") raise RuntimeError(f"Failed to get response for {url}")
if params.solve_cloudflare: if params.solve_cloudflare:
await self._solve_cloudflare(page_info.page) await self._cloudflare_solver(page_info.page)
# Make sure the page is fully loaded after the captcha # Make sure the page is fully loaded after the captcha
await self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle) await self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
+77 -255
View File
@@ -13,92 +13,55 @@ from patchright.sync_api import sync_playwright as sync_patchright
from patchright.async_api import async_playwright as async_patchright from patchright.async_api import async_playwright as async_patchright
from scrapling.core.utils import log from scrapling.core.utils import log
from scrapling.core._types import Unpack, TYPE_CHECKING
from ._types import PlaywrightSession, PlaywrightFetchParams
from ._base import SyncSession, AsyncSession, DynamicSessionMixin from ._base import SyncSession, AsyncSession, DynamicSessionMixin
from ._validators import validate_fetch as _validate, PlaywrightConfig from ._validators import validate_fetch as _validate, PlaywrightConfig
from scrapling.core._types import ( from scrapling.engines.toolbelt.convertor import Response, ResponseFactory
Any,
Dict,
List,
Optional,
Callable,
TYPE_CHECKING,
SelectorWaitStates,
)
from scrapling.engines.toolbelt.convertor import (
Response,
ResponseFactory,
)
from scrapling.engines.toolbelt.fingerprints import generate_convincing_referer from scrapling.engines.toolbelt.fingerprints import generate_convincing_referer
_UNSET: Any = object()
class DynamicSession(DynamicSessionMixin, SyncSession): class DynamicSession(DynamicSessionMixin, SyncSession):
"""A Browser session manager with page pooling.""" """A Browser session manager with page pooling."""
__slots__ = ( __slots__ = (
"max_pages", "_max_pages",
"headless", "_headless",
"hide_canvas", "_hide_canvas",
"disable_webgl", "_disable_webgl",
"real_chrome", "_real_chrome",
"stealth", "_stealth",
"google_search", "_google_search",
"proxy", "_proxy",
"locale", "_locale",
"extra_headers", "_extra_headers",
"useragent", "_useragent",
"timeout", "_timeout",
"cookies", "_cookies",
"disable_resources", "_disable_resources",
"network_idle", "_network_idle",
"load_dom", "_load_dom",
"wait_selector", "_wait_selector",
"init_script", "_init_script",
"wait_selector_state", "_wait_selector_state",
"wait", "_wait",
"playwright", "playwright",
"browser", "browser",
"context", "context",
"page_pool", "page_pool",
"_closed", "_closed",
"selector_config", "_selector_config",
"page_action", "_page_action",
"launch_options", "launch_options",
"context_options", "context_options",
"cdp_url", "_cdp_url",
"_headers_keys", "_headers_keys",
"_extra_flags",
"_additional_args",
"_user_data_dir",
) )
def __init__( def __init__(self, **kwargs: Unpack[PlaywrightSession]):
self,
__max_pages: int = 1,
headless: bool = True,
google_search: bool = True,
hide_canvas: bool = False,
disable_webgl: bool = False,
real_chrome: bool = False,
stealth: bool = False,
wait: int | float = 0,
page_action: Optional[Callable] = None,
proxy: Optional[str | Dict[str, str]] = None,
locale: str = "en-US",
extra_headers: Optional[Dict[str, str]] = None,
useragent: Optional[str] = None,
cdp_url: Optional[str] = None,
timeout: int | float = 30000,
disable_resources: bool = False,
wait_selector: Optional[str] = None,
init_script: Optional[str] = None,
cookies: Optional[List[Dict]] = None,
network_idle: bool = False,
load_dom: bool = True,
wait_selector_state: SelectorWaitStates = "attached",
user_data_dir: str = "",
extra_flags: Optional[List[str]] = None,
selector_config: Optional[Dict] = None,
additional_args: Optional[Dict] = None,
):
"""A Browser session manager with page pooling, it's using a persistent browser Context by default with a temporary user profile directory. """A Browser session manager with page pooling, it's using a persistent browser Context by default with a temporary user profile directory.
:param headless: Run the browser in headless/hidden (default), or headful/visible mode. :param headless: Run the browser in headless/hidden (default), or headful/visible mode.
@@ -129,105 +92,49 @@ class DynamicSession(DynamicSessionMixin, SyncSession):
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. :param selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings. :param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings.
""" """
self.__validate__( self.__validate__(**kwargs)
wait=wait, super().__init__(max_pages=self._max_pages)
proxy=proxy,
locale=locale,
timeout=timeout,
stealth=stealth,
cdp_url=cdp_url,
cookies=cookies,
load_dom=load_dom,
headless=headless,
useragent=useragent,
max_pages=__max_pages,
real_chrome=real_chrome,
page_action=page_action,
hide_canvas=hide_canvas,
init_script=init_script,
network_idle=network_idle,
user_data_dir=user_data_dir,
google_search=google_search,
extra_headers=extra_headers,
wait_selector=wait_selector,
disable_webgl=disable_webgl,
extra_flags=extra_flags,
selector_config=selector_config,
additional_args=additional_args,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
)
super().__init__(max_pages=self.max_pages)
def __create__(self): def __create__(self):
"""Create a browser for this instance and context.""" """Create a browser for this instance and context."""
sync_context = sync_patchright if self.stealth else sync_playwright sync_context = sync_patchright if self._stealth else sync_playwright
self.playwright: Playwright = sync_context().start() # pyright: ignore [reportAttributeAccessIssue] self.playwright: Playwright = sync_context().start() # pyright: ignore [reportAttributeAccessIssue]
if self.cdp_url: # pragma: no cover if self._cdp_url: # pragma: no cover
self.context = self.playwright.chromium.connect_over_cdp(endpoint_url=self.cdp_url).new_context( self.context = self.playwright.chromium.connect_over_cdp(endpoint_url=self._cdp_url).new_context(
**self.context_options **self.context_options
) )
else: else:
self.context = self.playwright.chromium.launch_persistent_context(**self.launch_options) self.context = self.playwright.chromium.launch_persistent_context(**self.launch_options)
if self.init_script: # pragma: no cover if self._init_script: # pragma: no cover
self.context.add_init_script(path=self.init_script) self.context.add_init_script(path=self._init_script)
if self.cookies: # pragma: no cover if self._cookies: # pragma: no cover
self.context.add_cookies(self.cookies) self.context.add_cookies(self._cookies)
def fetch( def fetch(self, url: str, **kwargs: Unpack[PlaywrightFetchParams]) -> Response:
self,
url: str,
google_search: bool = _UNSET,
timeout: int | float = _UNSET,
wait: int | float = _UNSET,
page_action: Optional[Callable] = _UNSET,
extra_headers: Optional[Dict[str, str]] = _UNSET,
disable_resources: bool = _UNSET,
wait_selector: Optional[str] = _UNSET,
wait_selector_state: SelectorWaitStates = _UNSET,
network_idle: bool = _UNSET,
load_dom: bool = _UNSET,
selector_config: Optional[Dict] = _UNSET,
) -> Response:
"""Opens up the browser and do your request based on your chosen options. """Opens up the browser and do your request based on your chosen options.
:param url: The Target url. :param url: The Target url.
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. :param kwargs: Additional keyword arguments including:
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - page_action: Added for automation. A function that takes the `page` object and does the automation you need.
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
- disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites.
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
:param wait_selector: Wait for a specific CSS selector to be in a specific state. - wait_selector: Wait for a specific CSS selector to be in a specific state.
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. - network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute.
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. - selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:return: A `Response` object. :return: A `Response` object.
""" """
params = _validate( params = _validate(kwargs, self, PlaywrightConfig)
[
("google_search", google_search, self.google_search),
("timeout", timeout, self.timeout),
("wait", wait, self.wait),
("page_action", page_action, self.page_action),
("extra_headers", extra_headers, self.extra_headers),
("disable_resources", disable_resources, self.disable_resources),
("wait_selector", wait_selector, self.wait_selector),
("wait_selector_state", wait_selector_state, self.wait_selector_state),
("network_idle", network_idle, self.network_idle),
("load_dom", load_dom, self.load_dom),
("selector_config", selector_config, self.selector_config),
],
PlaywrightConfig,
_UNSET,
)
if self._closed: # pragma: no cover if self._closed: # pragma: no cover
raise RuntimeError("Context manager has been closed") raise RuntimeError("Context manager has been closed")
@@ -285,35 +192,7 @@ class DynamicSession(DynamicSessionMixin, SyncSession):
class AsyncDynamicSession(DynamicSessionMixin, AsyncSession): class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
"""An async Browser session manager with page pooling, it's using a persistent browser Context by default with a temporary user profile directory.""" """An async Browser session manager with page pooling, it's using a persistent browser Context by default with a temporary user profile directory."""
def __init__( def __init__(self, **kwargs: Unpack[PlaywrightSession]):
self,
max_pages: int = 1,
headless: bool = True,
google_search: bool = True,
hide_canvas: bool = False,
disable_webgl: bool = False,
real_chrome: bool = False,
stealth: bool = False,
wait: int | float = 0,
page_action: Optional[Callable] = None,
proxy: Optional[str | Dict[str, str]] = None,
locale: str = "en-US",
extra_headers: Optional[Dict[str, str]] = None,
useragent: Optional[str] = None,
cdp_url: Optional[str] = None,
timeout: int | float = 30000,
disable_resources: bool = False,
wait_selector: Optional[str] = None,
init_script: Optional[str] = None,
cookies: Optional[List[Dict]] = None,
network_idle: bool = False,
load_dom: bool = True,
wait_selector_state: SelectorWaitStates = "attached",
user_data_dir: str = "",
extra_flags: Optional[List[str]] = None,
selector_config: Optional[Dict] = None,
additional_args: Optional[Dict] = None,
):
"""A Browser session manager with page pooling """A Browser session manager with page pooling
:param headless: Run the browser in headless/hidden (default), or headful/visible mode. :param headless: Run the browser in headless/hidden (default), or headful/visible mode.
@@ -345,107 +224,50 @@ class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. :param selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings. :param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings.
""" """
self.__validate__(**kwargs)
self.__validate__( super().__init__(max_pages=self._max_pages)
wait=wait,
proxy=proxy,
locale=locale,
timeout=timeout,
stealth=stealth,
cdp_url=cdp_url,
cookies=cookies,
load_dom=load_dom,
headless=headless,
useragent=useragent,
max_pages=max_pages,
real_chrome=real_chrome,
page_action=page_action,
hide_canvas=hide_canvas,
init_script=init_script,
network_idle=network_idle,
user_data_dir=user_data_dir,
google_search=google_search,
extra_headers=extra_headers,
wait_selector=wait_selector,
disable_webgl=disable_webgl,
extra_flags=extra_flags,
selector_config=selector_config,
additional_args=additional_args,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
)
super().__init__(max_pages=self.max_pages)
async def __create__(self): async def __create__(self):
"""Create a browser for this instance and context.""" """Create a browser for this instance and context."""
async_context = async_patchright if self.stealth else async_playwright async_context = async_patchright if self._stealth else async_playwright
self.playwright: AsyncPlaywright = await async_context().start() # pyright: ignore [reportAttributeAccessIssue] self.playwright: AsyncPlaywright = await async_context().start() # pyright: ignore [reportAttributeAccessIssue]
if self.cdp_url: if self._cdp_url:
browser = await self.playwright.chromium.connect_over_cdp(endpoint_url=self.cdp_url) browser = await self.playwright.chromium.connect_over_cdp(endpoint_url=self._cdp_url)
self.context: AsyncBrowserContext = await browser.new_context(**self.context_options) self.context: AsyncBrowserContext = await browser.new_context(**self.context_options)
else: else:
self.context: AsyncBrowserContext = await self.playwright.chromium.launch_persistent_context( self.context: AsyncBrowserContext = await self.playwright.chromium.launch_persistent_context(
**self.launch_options **self.launch_options
) )
if self.init_script: # pragma: no cover if self._init_script: # pragma: no cover
await self.context.add_init_script(path=self.init_script) await self.context.add_init_script(path=self._init_script)
if self.cookies: if self._cookies:
await self.context.add_cookies(self.cookies) # pyright: ignore await self.context.add_cookies(self._cookies) # pyright: ignore
async def fetch( async def fetch(self, url: str, **kwargs: Unpack[PlaywrightFetchParams]) -> Response:
self,
url: str,
google_search: bool = _UNSET,
timeout: int | float = _UNSET,
wait: int | float = _UNSET,
page_action: Optional[Callable] = _UNSET,
extra_headers: Optional[Dict[str, str]] = _UNSET,
disable_resources: bool = _UNSET,
wait_selector: Optional[str] = _UNSET,
wait_selector_state: SelectorWaitStates = _UNSET,
network_idle: bool = _UNSET,
load_dom: bool = _UNSET,
selector_config: Optional[Dict] = _UNSET,
) -> Response:
"""Opens up the browser and do your request based on your chosen options. """Opens up the browser and do your request based on your chosen options.
:param url: The Target url. :param url: The Target url.
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. :param kwargs: Additional keyword arguments including:
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - page_action: Added for automation. A function that takes the `page` object and does the automation you need.
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
- disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites.
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
:param wait_selector: Wait for a specific CSS selector to be in a specific state. - wait_selector: Wait for a specific CSS selector to be in a specific state.
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. - network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute.
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. - selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:return: A `Response` object. :return: A `Response` object.
""" """
params = _validate( params = _validate(kwargs, self, PlaywrightConfig)
[
("google_search", google_search, self.google_search),
("timeout", timeout, self.timeout),
("wait", wait, self.wait),
("page_action", page_action, self.page_action),
("extra_headers", extra_headers, self.extra_headers),
("disable_resources", disable_resources, self.disable_resources),
("wait_selector", wait_selector, self.wait_selector),
("wait_selector_state", wait_selector_state, self.wait_selector_state),
("network_idle", network_idle, self.network_idle),
("load_dom", load_dom, self.load_dom),
("selector_config", selector_config, self.selector_config),
],
PlaywrightConfig,
_UNSET,
)
if self._closed: # pragma: no cover if self._closed: # pragma: no cover
raise RuntimeError("Context manager has been closed") raise RuntimeError("Context manager has been closed")
+19 -8
View File
@@ -1,20 +1,21 @@
from threading import RLock from threading import RLock
from dataclasses import dataclass from dataclasses import dataclass
from playwright.sync_api import Page as SyncPage from playwright.sync_api._generated import Page as SyncPage
from playwright.async_api import Page as AsyncPage from playwright.async_api._generated import Page as AsyncPage
from scrapling.core._types import Optional, List, Literal from scrapling.core._types import Optional, List, Literal, overload, TypeVar, Generic, cast
PageState = Literal["ready", "busy", "error"] # States that a page can be in PageState = Literal["ready", "busy", "error"] # States that a page can be in
PageType = TypeVar("PageType", SyncPage, AsyncPage)
@dataclass @dataclass
class PageInfo: class PageInfo(Generic[PageType]):
"""Information about the page and its current state""" """Information about the page and its current state"""
__slots__ = ("page", "state", "url") __slots__ = ("page", "state", "url")
page: SyncPage | AsyncPage page: PageType
state: PageState state: PageState
url: Optional[str] url: Optional[str]
@@ -44,16 +45,26 @@ class PagePool:
def __init__(self, max_pages: int = 5): def __init__(self, max_pages: int = 5):
self.max_pages = max_pages self.max_pages = max_pages
self.pages: List[PageInfo] = [] self.pages: List[PageInfo[SyncPage] | PageInfo[AsyncPage]] = []
self._lock = RLock() self._lock = RLock()
def add_page(self, page: SyncPage | AsyncPage) -> PageInfo: @overload
def add_page(self, page: SyncPage) -> PageInfo[SyncPage]: ...
@overload
def add_page(self, page: AsyncPage) -> PageInfo[AsyncPage]: ...
def add_page(self, page: SyncPage | AsyncPage) -> PageInfo[SyncPage] | PageInfo[AsyncPage]:
"""Add a new page to the pool""" """Add a new page to the pool"""
with self._lock: with self._lock:
if len(self.pages) >= self.max_pages: if len(self.pages) >= self.max_pages:
raise RuntimeError(f"Maximum page limit ({self.max_pages}) reached") raise RuntimeError(f"Maximum page limit ({self.max_pages}) reached")
page_info = PageInfo(page, "ready", "") if isinstance(page, AsyncPage):
page_info = cast(PageInfo[AsyncPage], PageInfo(page, "ready", ""))
else:
page_info = cast(PageInfo[SyncPage], PageInfo(page, "ready", ""))
self.pages.append(page_info) self.pages.append(page_info)
return page_info return page_info
+65
View File
@@ -10,8 +10,11 @@ from scrapling.core._types import (
Tuple, Tuple,
Mapping, Mapping,
Optional, Optional,
Callable,
Iterable,
TypedDict, TypedDict,
TypeAlias, TypeAlias,
SelectorWaitStates,
TYPE_CHECKING, TYPE_CHECKING,
) )
@@ -49,7 +52,69 @@ if TYPE_CHECKING: # pragma: no cover
data: Optional[Dict | str] data: Optional[Dict | str]
json: Optional[Dict | List] json: Optional[Dict | List]
# Types for browser session
class BrowserSession(TypedDict, total=False):
max_pages: int
headless: bool
disable_resources: bool
network_idle: bool
load_dom: bool
wait_selector: Optional[str]
wait_selector_state: SelectorWaitStates
cookies: Optional[Iterable[Dict]]
google_search: bool
wait: int | float
page_action: Optional[Callable]
proxy: Optional[str | Dict[str, str] | Tuple]
extra_headers: Optional[Dict[str, str]]
timeout: int | float
init_script: Optional[str]
user_data_dir: str
selector_config: Optional[Dict]
additional_args: Optional[Dict]
class PlaywrightSession(BrowserSession, total=False):
cdp_url: Optional[str]
hide_canvas: bool
disable_webgl: bool
real_chrome: bool
stealth: bool
locale: str
useragent: Optional[str]
extra_flags: Optional[List[str]]
class PlaywrightFetchParams(TypedDict, total=False):
google_search: bool
timeout: int | float
wait: int | float
page_action: Optional[Callable]
extra_headers: Optional[Dict[str, str]]
disable_resources: bool
wait_selector: Optional[str]
wait_selector_state: SelectorWaitStates
network_idle: bool
load_dom: bool
selector_config: Optional[Dict]
class CamoufoxSession(BrowserSession, total=False):
block_images: bool
block_webrtc: bool
allow_webgl: bool
humanize: bool | float
solve_cloudflare: bool
addons: Optional[List[str]]
os_randomize: bool
disable_ads: bool
geoip: bool
class CamoufoxFetchParams(PlaywrightFetchParams, total=False):
solve_cloudflare: bool
else: # pragma: no cover else: # pragma: no cover
RequestsSession = TypedDict RequestsSession = TypedDict
GetRequestParams = TypedDict GetRequestParams = TypedDict
DataRequestParams = TypedDict DataRequestParams = TypedDict
PlaywrightSession = TypedDict
PlaywrightFetchParams = TypedDict
CamoufoxSession = TypedDict
CamoufoxFetchParams = TypedDict
+15 -5
View File
@@ -7,6 +7,7 @@ from dataclasses import dataclass, fields
from msgspec import Struct, Meta, convert, ValidationError from msgspec import Struct, Meta, convert, ValidationError
from scrapling.core._types import ( from scrapling.core._types import (
Any,
Dict, Dict,
List, List,
Tuple, Tuple,
@@ -17,6 +18,7 @@ from scrapling.core._types import (
overload, overload,
) )
from scrapling.engines.toolbelt.navigation import construct_proxy_dict from scrapling.engines.toolbelt.navigation import construct_proxy_dict
from scrapling.engines._browsers._types import PlaywrightFetchParams, CamoufoxFetchParams
# Custom validators for msgspec # Custom validators for msgspec
@@ -194,16 +196,24 @@ class _fetch_params:
def validate_fetch( def validate_fetch(
params: List[Tuple], model: type[PlaywrightConfig] | type[CamoufoxConfig], sentinel=None method_kwargs: Dict | PlaywrightFetchParams | CamoufoxFetchParams,
session: Any,
model: type[PlaywrightConfig] | type[CamoufoxConfig],
) -> _fetch_params: # pragma: no cover ) -> _fetch_params: # pragma: no cover
result = {} result = {}
overrides = {} overrides = {}
for arg, request_value, session_value in params: # Get all field names that _fetch_params needs
if request_value is not sentinel: fetch_param_fields = {f.name for f in fields(_fetch_params)}
overrides[arg] = request_value
for key in fetch_param_fields:
if key in method_kwargs:
overrides[key] = method_kwargs[key]
else: else:
result[arg] = session_value # Check for underscore-prefixed attribute (private)
attr_name = f"_{key}"
if hasattr(session, attr_name):
result[key] = getattr(session, attr_name)
if overrides: if overrides:
validated_config = validate(overrides, model) validated_config = validate(overrides, model)