f0b01fc253
Made the code shorter by an additional ~200 lines and easier to maintain in return for making the arguments autocompletion bad for shells that don't check for dynamic type hints like IPython.
432 lines
17 KiB
Python
432 lines
17 KiB
Python
from time import time
|
|
from asyncio import sleep as asyncio_sleep, Lock
|
|
|
|
from camoufox import DefaultAddons
|
|
from playwright.sync_api._generated import Page
|
|
from playwright.sync_api import (
|
|
Frame,
|
|
BrowserContext,
|
|
Playwright,
|
|
Response as SyncPlaywrightResponse,
|
|
)
|
|
from playwright.async_api._generated import Page as AsyncPage
|
|
from playwright.async_api import (
|
|
Frame as AsyncFrame,
|
|
Playwright as AsyncPlaywright,
|
|
Response as AsyncPlaywrightResponse,
|
|
BrowserContext as AsyncBrowserContext,
|
|
)
|
|
from playwright._impl._errors import Error as PlaywrightError
|
|
from camoufox.pkgman import installed_verstr as camoufox_version
|
|
from camoufox.utils import launch_options as generate_launch_options
|
|
|
|
from ._page import PageInfo, PagePool
|
|
from scrapling.parser import Selector
|
|
from scrapling.core._types import Any, cast, Dict, List, Optional, Callable, TYPE_CHECKING
|
|
from scrapling.engines.toolbelt.fingerprints import get_os_name
|
|
from ._validators import validate, PlaywrightConfig, CamoufoxConfig
|
|
from ._config_tools import _compiled_stealth_scripts, _launch_kwargs, _context_kwargs
|
|
from scrapling.engines.toolbelt.navigation import intercept_route, async_intercept_route
|
|
|
|
__ff_version_str__ = camoufox_version().split(".", 1)[0]
|
|
|
|
|
|
class SyncSession:
|
|
def __init__(self, max_pages: int = 1):
|
|
self.max_pages = max_pages
|
|
self.page_pool = PagePool(max_pages)
|
|
self._max_wait_for_page = 60
|
|
self.playwright: Playwright | Any = None
|
|
self.context: BrowserContext | Any = None
|
|
self._closed = False
|
|
|
|
def __create__(self):
|
|
pass
|
|
|
|
def close(self): # pragma: no cover
|
|
"""Close all resources"""
|
|
if self._closed:
|
|
return
|
|
|
|
if self.context:
|
|
self.context.close()
|
|
self.context = None
|
|
|
|
if self.playwright:
|
|
self.playwright.stop()
|
|
self.playwright = None # pyright: ignore
|
|
|
|
self._closed = True
|
|
|
|
def __enter__(self):
|
|
self.__create__()
|
|
return self
|
|
|
|
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
self.close()
|
|
|
|
def _get_page(
|
|
self,
|
|
timeout: int | float,
|
|
extra_headers: Optional[Dict[str, str]],
|
|
disable_resources: bool,
|
|
) -> PageInfo[Page]: # pragma: no cover
|
|
"""Get a new page to use"""
|
|
|
|
# No need to check if a page is available or not in sync code because the code blocked before reaching here till the page closed, ofc.
|
|
assert self.context is not None, "Browser context not initialized"
|
|
page = self.context.new_page()
|
|
page.set_default_navigation_timeout(timeout)
|
|
page.set_default_timeout(timeout)
|
|
if extra_headers:
|
|
page.set_extra_http_headers(extra_headers)
|
|
|
|
if disable_resources:
|
|
page.route("**/*", intercept_route)
|
|
|
|
if getattr(self, "stealth", False):
|
|
for script in _compiled_stealth_scripts():
|
|
page.add_init_script(script=script)
|
|
|
|
page_info = self.page_pool.add_page(page)
|
|
page_info.mark_busy()
|
|
return page_info
|
|
|
|
def get_pool_stats(self) -> Dict[str, int]:
|
|
"""Get statistics about the current page pool"""
|
|
return {
|
|
"total_pages": self.page_pool.pages_count,
|
|
"busy_pages": self.page_pool.busy_count,
|
|
"max_pages": self.max_pages,
|
|
}
|
|
|
|
@staticmethod
|
|
def _wait_for_networkidle(page: Page | Frame, timeout: Optional[int] = None):
|
|
"""Wait for the page to become idle (no network activity) even if there are never-ending requests."""
|
|
try:
|
|
page.wait_for_load_state("networkidle", timeout=timeout)
|
|
except PlaywrightError:
|
|
pass
|
|
|
|
def _wait_for_page_stability(self, page: Page | Frame, load_dom: bool, network_idle: bool):
|
|
page.wait_for_load_state(state="load")
|
|
if load_dom:
|
|
page.wait_for_load_state(state="domcontentloaded")
|
|
if network_idle:
|
|
self._wait_for_networkidle(page)
|
|
|
|
@staticmethod
|
|
def _create_response_handler(page_info: PageInfo[Page], response_container: List) -> Callable:
|
|
"""Create a response handler that captures the final navigation response.
|
|
|
|
:param page_info: The PageInfo object containing the page
|
|
:param response_container: A list to store the final response (mutable container)
|
|
:return: A callback function for page.on("response", ...)
|
|
"""
|
|
|
|
def handle_response(finished_response: SyncPlaywrightResponse):
|
|
if (
|
|
finished_response.request.resource_type == "document"
|
|
and finished_response.request.is_navigation_request()
|
|
and finished_response.request.frame == page_info.page.main_frame
|
|
):
|
|
response_container[0] = finished_response
|
|
|
|
return handle_response
|
|
|
|
|
|
class AsyncSession:
|
|
def __init__(self, max_pages: int = 1):
|
|
self.max_pages = max_pages
|
|
self.page_pool = PagePool(max_pages)
|
|
self._max_wait_for_page = 60
|
|
self.playwright: AsyncPlaywright | Any = None
|
|
self.context: AsyncBrowserContext | Any = None
|
|
self._closed = False
|
|
self._lock = Lock()
|
|
|
|
async def __create__(self):
|
|
pass
|
|
|
|
async def close(self):
|
|
"""Close all resources"""
|
|
if self._closed: # pragma: no cover
|
|
return
|
|
|
|
if self.context:
|
|
await self.context.close()
|
|
self.context = None # pyright: ignore
|
|
|
|
if self.playwright:
|
|
await self.playwright.stop()
|
|
self.playwright = None # pyright: ignore
|
|
|
|
self._closed = True
|
|
|
|
async def __aenter__(self):
|
|
await self.__create__()
|
|
return self
|
|
|
|
async def __aexit__(self, exc_type, exc_val, exc_tb):
|
|
await self.close()
|
|
|
|
async def _get_page(
|
|
self,
|
|
timeout: int | float,
|
|
extra_headers: Optional[Dict[str, str]],
|
|
disable_resources: bool,
|
|
) -> PageInfo[AsyncPage]: # pragma: no cover
|
|
"""Get a new page to use"""
|
|
if TYPE_CHECKING:
|
|
assert self.context is not None, "Browser context not initialized"
|
|
|
|
async with self._lock:
|
|
# If we're at max capacity after cleanup, wait for busy pages to finish
|
|
if self.page_pool.pages_count >= self.max_pages:
|
|
start_time = time()
|
|
while time() - start_time < self._max_wait_for_page:
|
|
await asyncio_sleep(0.05)
|
|
if self.page_pool.pages_count < self.max_pages:
|
|
break
|
|
else:
|
|
raise TimeoutError(
|
|
f"No pages finished to clear place in the pool within the {self._max_wait_for_page}s timeout period"
|
|
)
|
|
|
|
page = await self.context.new_page()
|
|
page.set_default_navigation_timeout(timeout)
|
|
page.set_default_timeout(timeout)
|
|
if extra_headers:
|
|
await page.set_extra_http_headers(extra_headers)
|
|
|
|
if disable_resources:
|
|
await page.route("**/*", async_intercept_route)
|
|
|
|
if getattr(self, "stealth", False):
|
|
for script in _compiled_stealth_scripts():
|
|
await page.add_init_script(script=script)
|
|
|
|
return self.page_pool.add_page(page)
|
|
|
|
def get_pool_stats(self) -> Dict[str, int]:
|
|
"""Get statistics about the current page pool"""
|
|
return {
|
|
"total_pages": self.page_pool.pages_count,
|
|
"busy_pages": self.page_pool.busy_count,
|
|
"max_pages": self.max_pages,
|
|
}
|
|
|
|
@staticmethod
|
|
async def _wait_for_networkidle(page: AsyncPage | AsyncFrame, timeout: Optional[int] = None):
|
|
"""Wait for the page to become idle (no network activity) even if there are never-ending requests."""
|
|
try:
|
|
await page.wait_for_load_state("networkidle", timeout=timeout)
|
|
except PlaywrightError:
|
|
pass
|
|
|
|
async def _wait_for_page_stability(self, page: AsyncPage | AsyncFrame, load_dom: bool, network_idle: bool):
|
|
await page.wait_for_load_state(state="load")
|
|
if load_dom:
|
|
await page.wait_for_load_state(state="domcontentloaded")
|
|
if network_idle:
|
|
await self._wait_for_networkidle(page)
|
|
|
|
@staticmethod
|
|
def _create_response_handler(page_info: PageInfo[AsyncPage], response_container: List) -> Callable:
|
|
"""Create an async response handler that captures the final navigation response.
|
|
|
|
:param page_info: The PageInfo object containing the page
|
|
:param response_container: A list to store the final response (mutable container)
|
|
:return: A callback function for page.on("response", ...)
|
|
"""
|
|
|
|
async def handle_response(finished_response: AsyncPlaywrightResponse):
|
|
if (
|
|
finished_response.request.resource_type == "document"
|
|
and finished_response.request.is_navigation_request()
|
|
and finished_response.request.frame == page_info.page.main_frame
|
|
):
|
|
response_container[0] = finished_response
|
|
|
|
return handle_response
|
|
|
|
|
|
class DynamicSessionMixin:
|
|
def __validate__(self, **params):
|
|
if "__max_pages" in params:
|
|
params["max_pages"] = params.pop("__max_pages")
|
|
|
|
config = validate(params, model=PlaywrightConfig)
|
|
|
|
self._max_pages = config.max_pages
|
|
self._headless = config.headless
|
|
self._hide_canvas = config.hide_canvas
|
|
self._disable_webgl = config.disable_webgl
|
|
self._real_chrome = config.real_chrome
|
|
self._stealth = config.stealth
|
|
self._google_search = config.google_search
|
|
self._wait = config.wait
|
|
self._proxy = config.proxy
|
|
self._locale = config.locale
|
|
self._extra_headers = config.extra_headers
|
|
self._useragent = config.useragent
|
|
self._timeout = config.timeout
|
|
self._cookies = config.cookies
|
|
self._disable_resources = config.disable_resources
|
|
self._cdp_url = config.cdp_url
|
|
self._network_idle = config.network_idle
|
|
self._load_dom = config.load_dom
|
|
self._wait_selector = config.wait_selector
|
|
self._init_script = config.init_script
|
|
self._wait_selector_state = config.wait_selector_state
|
|
self._extra_flags = config.extra_flags
|
|
self._selector_config = config.selector_config
|
|
self._additional_args = config.additional_args
|
|
self._page_action = config.page_action
|
|
self._user_data_dir = config.user_data_dir
|
|
self._headers_keys = {header.lower() for header in self._extra_headers.keys()} if self._extra_headers else set()
|
|
self.__initiate_browser_options__()
|
|
|
|
def __initiate_browser_options__(self):
|
|
if TYPE_CHECKING:
|
|
assert isinstance(self._proxy, tuple)
|
|
|
|
if not self._cdp_url:
|
|
# `launch_options` is used with persistent context
|
|
self.launch_options = dict(
|
|
_launch_kwargs(
|
|
self._headless,
|
|
self._proxy,
|
|
self._locale,
|
|
tuple(self._extra_headers.items()) if self._extra_headers else tuple(),
|
|
self._useragent,
|
|
self._real_chrome,
|
|
self._stealth,
|
|
self._hide_canvas,
|
|
self._disable_webgl,
|
|
tuple(self._extra_flags) if self._extra_flags else tuple(),
|
|
)
|
|
)
|
|
self.launch_options["extra_http_headers"] = dict(self.launch_options["extra_http_headers"])
|
|
self.launch_options["proxy"] = dict(self.launch_options["proxy"]) or None
|
|
self.launch_options["user_data_dir"] = self._user_data_dir
|
|
self.launch_options.update(cast(Dict, self._additional_args))
|
|
self.context_options = dict()
|
|
else:
|
|
# while `context_options` is left to be used when cdp mode is enabled
|
|
self.launch_options = dict()
|
|
self.context_options = dict(
|
|
_context_kwargs(
|
|
self._proxy,
|
|
self._locale,
|
|
tuple(self._extra_headers.items()) if self._extra_headers else tuple(),
|
|
self._useragent,
|
|
self._stealth,
|
|
)
|
|
)
|
|
self.context_options["extra_http_headers"] = dict(self.context_options["extra_http_headers"])
|
|
self.context_options["proxy"] = dict(self.context_options["proxy"]) or None
|
|
self.context_options.update(cast(Dict, self._additional_args))
|
|
|
|
|
|
class StealthySessionMixin:
|
|
def __validate__(self, **params):
|
|
if "__max_pages" in params:
|
|
params["max_pages"] = params.pop("__max_pages")
|
|
|
|
config: CamoufoxConfig = validate(params, model=CamoufoxConfig)
|
|
|
|
self._max_pages = config.max_pages
|
|
self._headless = config.headless
|
|
self._block_images = config.block_images
|
|
self._disable_resources = config.disable_resources
|
|
self._block_webrtc = config.block_webrtc
|
|
self._allow_webgl = config.allow_webgl
|
|
self._network_idle = config.network_idle
|
|
self._load_dom = config.load_dom
|
|
self._humanize = config.humanize
|
|
self._solve_cloudflare = config.solve_cloudflare
|
|
self._wait = config.wait
|
|
self._timeout = config.timeout
|
|
self._page_action = config.page_action
|
|
self._wait_selector = config.wait_selector
|
|
self._init_script = config.init_script
|
|
self._addons = config.addons
|
|
self._wait_selector_state = config.wait_selector_state
|
|
self._cookies = config.cookies
|
|
self._google_search = config.google_search
|
|
self._extra_headers = config.extra_headers
|
|
self._proxy = config.proxy
|
|
self._os_randomize = config.os_randomize
|
|
self._disable_ads = config.disable_ads
|
|
self._geoip = config.geoip
|
|
self._selector_config = config.selector_config
|
|
self._additional_args = config.additional_args
|
|
self._user_data_dir = config.user_data_dir
|
|
self._headers_keys = {header.lower() for header in self._extra_headers.keys()} if self._extra_headers else set()
|
|
self.__initiate_browser_options__()
|
|
|
|
def __initiate_browser_options__(self):
|
|
"""Initiate browser options."""
|
|
self.launch_options: Dict[str, Any] = generate_launch_options(
|
|
**{
|
|
"geoip": self._geoip,
|
|
"proxy": dict(self._proxy) if self._proxy and isinstance(self._proxy, tuple) else self._proxy,
|
|
"addons": self._addons,
|
|
"exclude_addons": [] if self._disable_ads else [DefaultAddons.UBO],
|
|
"headless": self._headless,
|
|
"humanize": True if self._solve_cloudflare else self._humanize,
|
|
"i_know_what_im_doing": True, # To turn warnings off with the user configurations
|
|
"allow_webgl": self._allow_webgl,
|
|
"block_webrtc": self._block_webrtc,
|
|
"block_images": self._block_images, # Careful! it makes some websites don't finish loading at all like stackoverflow even in headful mode.
|
|
"os": None if self._os_randomize else get_os_name(),
|
|
"user_data_dir": self._user_data_dir,
|
|
"ff_version": __ff_version_str__,
|
|
"firefox_user_prefs": {
|
|
# This is what enabling `enable_cache` does internally, so we do it from here instead
|
|
"browser.sessionhistory.max_entries": 10,
|
|
"browser.sessionhistory.max_total_viewers": -1,
|
|
"browser.cache.memory.enable": True,
|
|
"browser.cache.disk_cache_ssl": True,
|
|
"browser.cache.disk.smart_size.enabled": True,
|
|
},
|
|
**cast(Dict, self._additional_args),
|
|
}
|
|
)
|
|
|
|
@staticmethod
|
|
def _detect_cloudflare(page_content: str) -> str | None:
|
|
"""
|
|
Detect the type of Cloudflare challenge present in the provided page content.
|
|
|
|
This function analyzes the given page content to identify whether a specific
|
|
type of Cloudflare challenge is present. It checks for three predefined
|
|
challenge types: non-interactive, managed, and interactive. If a challenge
|
|
type is detected, it returns the corresponding type as a string. If no
|
|
challenge type is detected, it returns None.
|
|
|
|
Args:
|
|
page_content (str): The content of the page to analyze for Cloudflare
|
|
challenge types.
|
|
|
|
Returns:
|
|
str: A string representing the detected Cloudflare challenge type, if
|
|
found. Returns None if no challenge matches.
|
|
"""
|
|
challenge_types = (
|
|
"non-interactive",
|
|
"managed",
|
|
"interactive",
|
|
)
|
|
for ctype in challenge_types:
|
|
if f"cType: '{ctype}'" in page_content:
|
|
return ctype
|
|
|
|
# Check if turnstile captcha is embedded inside the page (Usually inside a closed Shadow iframe)
|
|
selector = Selector(content=page_content)
|
|
if selector.css('script[src*="challenges.cloudflare.com/turnstile/v"]'):
|
|
return "embedded"
|
|
|
|
return None
|