diff --git a/.github/workflows/docker-build.yml b/.github/workflows/docker-build.yml
index 62338f3..3c8ddcd 100644
--- a/.github/workflows/docker-build.yml
+++ b/.github/workflows/docker-build.yml
@@ -13,8 +13,8 @@ on:
default: 'latest'
env:
- REGISTRY: docker.io
- IMAGE_NAME: pyd4vinci/scrapling
+ DOCKERHUB_IMAGE: pyd4vinci/scrapling
+ GHCR_IMAGE: ghcr.io/${{ github.repository_owner }}/scrapling
jobs:
build-and-push:
@@ -35,15 +35,24 @@ jobs:
- name: Log in to Docker Hub
uses: docker/login-action@v3
with:
- registry: ${{ env.REGISTRY }}
+ registry: docker.io
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
+ - name: Log in to GitHub Container Registry
+ uses: docker/login-action@v3
+ with:
+ registry: ghcr.io
+ username: ${{ github.actor }}
+ password: ${{ secrets.CONTAINER_TOKEN }}
+
- name: Extract metadata
id: meta
uses: docker/metadata-action@v5
with:
- images: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}
+ images: |
+ ${{ env.DOCKERHUB_IMAGE }}
+ ${{ env.GHCR_IMAGE }}
tags: |
type=ref,event=branch
type=ref,event=pr
@@ -51,6 +60,14 @@ jobs:
type=semver,pattern={{major}}.{{minor}}
type=semver,pattern={{major}}
type=raw,value=latest,enable={{is_default_branch}}
+ labels: |
+ org.opencontainers.image.title=Scrapling
+ org.opencontainers.image.description=An undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!
+ org.opencontainers.image.vendor=D4Vinci
+ org.opencontainers.image.licenses=BSD
+ org.opencontainers.image.url=https://scrapling.readthedocs.io/en/latest/
+ org.opencontainers.image.source=${{ github.server_url }}/${{ github.repository }}
+ org.opencontainers.image.documentation=https://scrapling.readthedocs.io/en/latest/
- name: Build and push Docker image
uses: docker/build-push-action@v5
diff --git a/docs/fetching/dynamic.md b/docs/fetching/dynamic.md
index 3075fdc..fbdb387 100644
--- a/docs/fetching/dynamic.md
+++ b/docs/fetching/dynamic.md
@@ -89,6 +89,7 @@ Scrapling provides many options with this fetcher and its session classes. To ma
| locale | Set the locale for the browser if wanted. The default value is `en-US`. | ✔️ |
| cdp_url | Instead of launching a new browser instance, connect to this CDP URL to control real browsers through CDP. | ✔️ |
| user_data_dir | Path to a User Data Directory, which stores browser session data like cookies and local storage. The default is to create a temporary directory. **Only Works with sessions** | ✔️ |
+| extra_flags | A list of additional browser flags to pass to the browser on launch. | ✔️ |
| additional_args | Additional arguments to be passed to Playwright's context as additional settings, and they take higher priority than Scrapling's settings. | ✔️ |
| selector_config | A dictionary of custom parsing arguments to be used when creating the final `Selector`/`Response` class. | ✔️ |
@@ -300,4 +301,4 @@ Use DynamicFetcher when:
- Need custom browser config
- Want flexible stealth options
-If you want more stealth and control without much config, check out the [StealthyFetcher](stealthy.md).
\ No newline at end of file
+If you want more stealth and control without much config, check out the [StealthyFetcher](stealthy.md).
diff --git a/docs/tutorials/external.md b/docs/tutorials/external.md
new file mode 100644
index 0000000..15a5f9b
--- /dev/null
+++ b/docs/tutorials/external.md
@@ -0,0 +1,34 @@
+
+If you have issues with the browser installation, such as resource management, we recommend you try the Cloud Browser from [Scrapeless](https://www.scrapeless.com/en/product/scraping-browser?utm_source=official&utm_term=scrapling) for free!
+
+The usage is straightforward: create an account and [get your API key](https://docs.scrapeless.com/en/scraping-browser/quickstart/getting-started/?utm_source=official&utm_term=scrapling), then pass it to the `DynamicSession` like this:
+
+```python
+from urllib.parse import urlencode
+
+from scrapling.fetchers import DynamicSession
+
+# Configure your browser session
+config = {
+ "token": "YOUR_API_KEY",
+ "sessionName": "scrapling-session",
+ "sessionTTL": "300", # 5 minutes
+ "proxyCountry": "ANY",
+ "sessionRecording": "false",
+}
+
+# Build WebSocket URL
+ws_endpoint = f"wss://browser.scrapeless.com/api/v2/browser?{urlencode(config)}"
+print('Connecting to Scrapeless...')
+
+with DynamicSession(cdp_url=ws_endpoint, disable_resources=True) as s:
+ print("Connected!")
+ page = s.fetch("https://httpbin.org/headers", network_idle=True)
+ print(f"Page loaded, content length: {len(page.body)}")
+ print(page.json())
+```
+The `DynamicSession` class instance will work as usual, so no further explanation is needed.
+
+However, the Scrapeless Cloud Browser can be configured with proxy options, like the proxy country in the config above, [custom fingerprint](https://docs.scrapeless.com/en/scraping-browser/features/advanced-privacy-anti-detection/custom-fingerprint/?utm_source=official&utm_term=scrapling) configuration, [captcha solving](https://docs.scrapeless.com/en/scraping-browser/features/advanced-privacy-anti-detection/supported-captchas/?utm_source=official&utm_term=scrapling), and more.
+
+Check out the [Scrapeless's browser documentation](https://docs.scrapeless.com/en/scraping-browser/quickstart/introduction/?utm_source=official&utm_term=scrapling) for more details.
\ No newline at end of file
diff --git a/mkdocs.yml b/mkdocs.yml
index d455c4a..6a16483 100644
--- a/mkdocs.yml
+++ b/mkdocs.yml
@@ -83,6 +83,7 @@ nav:
- Tutorials:
- A Free Alternative to AI for Robust Web Scraping: tutorials/replacing_ai.md
- Migrating from BeautifulSoup: tutorials/migrating_from_beautifulsoup.md
+ - Using Scrapeless browser: tutorials/external.md
# - Migrating from AutoScraper: tutorials/migrating_from_autoscraper.md
- Development:
- API Reference:
diff --git a/pyproject.toml b/pyproject.toml
index 252f05d..2dbbb92 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -4,8 +4,8 @@ build-backend = "setuptools.build_meta"
[project]
name = "scrapling"
-# Static version instead of dynamic version so we can get better layer caching while building docker, check the docker file to understand
-version = "0.3.7"
+# Static version instead of a dynamic version so we can get better layer caching while building docker, check the docker file to understand
+version = "0.3.8"
description = "Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!"
readme = {file = "README.md", content-type = "text/markdown"}
license = {file = "LICENSE"}
@@ -59,7 +59,7 @@ classifiers = [
dependencies = [
"lxml>=6.0.2",
"cssselect>=1.3.0",
- "orjson>=3.11.3",
+ "orjson>=3.11.4",
"tldextract>=5.3.0",
]
@@ -74,7 +74,7 @@ fetchers = [
"msgspec>=0.19.0",
]
ai = [
- "mcp>=1.16.0",
+ "mcp>=1.19.0",
"markdownify>=1.2.0",
"scrapling[fetchers]",
]
@@ -89,6 +89,7 @@ all = [
[project.urls]
Homepage = "https://github.com/D4Vinci/Scrapling"
+Changelog = "https://github.com/D4Vinci/Scrapling/releases"
Documentation = "https://scrapling.readthedocs.io/en/latest/"
Repository = "https://github.com/D4Vinci/Scrapling"
"Bug Tracker" = "https://github.com/D4Vinci/Scrapling/issues"
diff --git a/scrapling/__init__.py b/scrapling/__init__.py
index cbc3a2c..2122396 100644
--- a/scrapling/__init__.py
+++ b/scrapling/__init__.py
@@ -1,5 +1,5 @@
__author__ = "Karim Shoair (karim.shoair@pm.me)"
-__version__ = "0.3.7"
+__version__ = "0.3.8"
__copyright__ = "Copyright (c) 2024 Karim Shoair"
from typing import Any, TYPE_CHECKING
diff --git a/scrapling/engines/_browsers/_base.py b/scrapling/engines/_browsers/_base.py
index 9a6dfae..f670e4c 100644
--- a/scrapling/engines/_browsers/_base.py
+++ b/scrapling/engines/_browsers/_base.py
@@ -2,17 +2,27 @@ from time import time
from asyncio import sleep as asyncio_sleep, Lock
from camoufox import DefaultAddons
-from playwright.sync_api import BrowserContext, Playwright
-from playwright.async_api import (
- BrowserContext as AsyncBrowserContext,
- Playwright as AsyncPlaywright,
+from playwright.sync_api import (
+ Page,
+ Frame,
+ BrowserContext,
+ Playwright,
+ Response as SyncPlaywrightResponse,
)
+from playwright.async_api import (
+ Page as AsyncPage,
+ Frame as AsyncFrame,
+ Playwright as AsyncPlaywright,
+ Response as AsyncPlaywrightResponse,
+ BrowserContext as AsyncBrowserContext,
+)
+from playwright._impl._errors import Error as PlaywrightError
from camoufox.pkgman import installed_verstr as camoufox_version
from camoufox.utils import launch_options as generate_launch_options
from ._page import PageInfo, PagePool
from scrapling.parser import Selector
-from scrapling.core._types import Any, cast, Dict, Optional, TYPE_CHECKING
+from scrapling.core._types import Any, cast, Dict, List, Optional, Callable, TYPE_CHECKING
from scrapling.engines.toolbelt.fingerprints import get_os_name
from ._validators import validate, PlaywrightConfig, CamoufoxConfig
from ._config_tools import _compiled_stealth_scripts, _launch_kwargs, _context_kwargs
@@ -26,10 +36,35 @@ class SyncSession:
self.max_pages = max_pages
self.page_pool = PagePool(max_pages)
self._max_wait_for_page = 60
- self.playwright: Optional[Playwright] = None
- self.context: Optional[BrowserContext] = None
+ self.playwright: Playwright | Any = None
+ self.context: BrowserContext | Any = None
self._closed = False
+ def __create__(self):
+ pass
+
+ def close(self): # pragma: no cover
+ """Close all resources"""
+ if self._closed:
+ return
+
+ if self.context:
+ self.context.close()
+ self.context = None
+
+ if self.playwright:
+ self.playwright.stop()
+ self.playwright = None # pyright: ignore
+
+ self._closed = True
+
+ def __enter__(self):
+ self.__create__()
+ return self
+
+ def __exit__(self, exc_type, exc_val, exc_tb):
+ self.close()
+
def _get_page(
self,
timeout: int | float,
@@ -53,7 +88,9 @@ class SyncSession:
for script in _compiled_stealth_scripts():
page.add_init_script(script=script)
- return self.page_pool.add_page(page)
+ page_info = self.page_pool.add_page(page)
+ page_info.mark_busy()
+ return page_info
def get_pool_stats(self) -> Dict[str, int]:
"""Get statistics about the current page pool"""
@@ -63,17 +100,76 @@ class SyncSession:
"max_pages": self.max_pages,
}
+ @staticmethod
+ def _wait_for_networkidle(page: Page | Frame, timeout: Optional[int] = None):
+ """Wait for the page to become idle (no network activity) even if there are never-ending requests."""
+ try:
+ page.wait_for_load_state("networkidle", timeout=timeout)
+ except PlaywrightError:
+ pass
+
+ def _wait_for_page_stability(self, page: Page | Frame, load_dom: bool, network_idle: bool):
+ page.wait_for_load_state(state="load")
+ if load_dom:
+ page.wait_for_load_state(state="domcontentloaded")
+ if network_idle:
+ self._wait_for_networkidle(page)
+
+ @staticmethod
+ def _create_response_handler(page_info: PageInfo, response_container: List) -> Callable:
+ """Create a response handler that captures the final navigation response.
+
+ :param page_info: The PageInfo object containing the page
+ :param response_container: A list to store the final response (mutable container)
+ :return: A callback function for page.on("response", ...)
+ """
+
+ def handle_response(finished_response: SyncPlaywrightResponse):
+ if (
+ finished_response.request.resource_type == "document"
+ and finished_response.request.is_navigation_request()
+ and finished_response.request.frame == page_info.page.main_frame
+ ):
+ response_container[0] = finished_response
+
+ return handle_response
+
class AsyncSession:
def __init__(self, max_pages: int = 1):
self.max_pages = max_pages
self.page_pool = PagePool(max_pages)
self._max_wait_for_page = 60
- self.playwright: Optional[AsyncPlaywright] = None
- self.context: Optional[AsyncBrowserContext] = None
+ self.playwright: AsyncPlaywright | Any = None
+ self.context: AsyncBrowserContext | Any = None
self._closed = False
self._lock = Lock()
+ async def __create__(self):
+ pass
+
+ async def close(self):
+ """Close all resources"""
+ if self._closed: # pragma: no cover
+ return
+
+ if self.context:
+ await self.context.close()
+ self.context = None # pyright: ignore
+
+ if self.playwright:
+ await self.playwright.stop()
+ self.playwright = None # pyright: ignore
+
+ self._closed = True
+
+ async def __aenter__(self):
+ await self.__create__()
+ return self
+
+ async def __aexit__(self, exc_type, exc_val, exc_tb):
+ await self.close()
+
async def _get_page(
self,
timeout: int | float,
@@ -97,7 +193,6 @@ class AsyncSession:
f"No pages finished to clear place in the pool within the {self._max_wait_for_page}s timeout period"
)
- assert self.context is not None, "Browser context not initialized"
page = await self.context.new_page()
page.set_default_navigation_timeout(timeout)
page.set_default_timeout(timeout)
@@ -121,6 +216,40 @@ class AsyncSession:
"max_pages": self.max_pages,
}
+ @staticmethod
+ async def _wait_for_networkidle(page: AsyncPage | AsyncFrame, timeout: Optional[int] = None):
+ """Wait for the page to become idle (no network activity) even if there are never-ending requests."""
+ try:
+ await page.wait_for_load_state("networkidle", timeout=timeout)
+ except PlaywrightError:
+ pass
+
+ async def _wait_for_page_stability(self, page: AsyncPage | AsyncFrame, load_dom: bool, network_idle: bool):
+ await page.wait_for_load_state(state="load")
+ if load_dom:
+ await page.wait_for_load_state(state="domcontentloaded")
+ if network_idle:
+ await self._wait_for_networkidle(page)
+
+ @staticmethod
+ def _create_response_handler(page_info: PageInfo, response_container: List) -> Callable:
+ """Create an async response handler that captures the final navigation response.
+
+ :param page_info: The PageInfo object containing the page
+ :param response_container: A list to store the final response (mutable container)
+ :return: A callback function for page.on("response", ...)
+ """
+
+ async def handle_response(finished_response: AsyncPlaywrightResponse):
+ if (
+ finished_response.request.resource_type == "document"
+ and finished_response.request.is_navigation_request()
+ and finished_response.request.frame == page_info.page.main_frame
+ ):
+ response_container[0] = finished_response
+
+ return handle_response
+
class DynamicSessionMixin:
def __validate__(self, **params):
@@ -147,6 +276,7 @@ class DynamicSessionMixin:
self.wait_selector = config.wait_selector
self.init_script = config.init_script
self.wait_selector_state = config.wait_selector_state
+ self.extra_flags = config.extra_flags
self.selector_config = config.selector_config
self.additional_args = config.additional_args
self.page_action = config.page_action
@@ -171,6 +301,7 @@ class DynamicSessionMixin:
self.stealth,
self.hide_canvas,
self.disable_webgl,
+ tuple(self.extra_flags) if self.extra_flags else tuple(),
)
)
self.launch_options["extra_http_headers"] = dict(self.launch_options["extra_http_headers"])
diff --git a/scrapling/engines/_browsers/_camoufox.py b/scrapling/engines/_browsers/_camoufox.py
index 94e8d37..4ac0558 100644
--- a/scrapling/engines/_browsers/_camoufox.py
+++ b/scrapling/engines/_browsers/_camoufox.py
@@ -2,22 +2,19 @@ from random import randint
from re import compile as re_compile
from playwright.sync_api import (
- Response as SyncPlaywrightResponse,
- sync_playwright,
- Locator,
Page,
+ Locator,
+ sync_playwright,
)
from playwright.async_api import (
async_playwright,
- Response as AsyncPlaywrightResponse,
- BrowserContext as AsyncBrowserContext,
- Playwright as AsyncPlaywright,
- Locator as AsyncLocator,
Page as async_Page,
+ Locator as AsyncLocator,
+ Playwright as AsyncPlaywright,
+ BrowserContext as AsyncBrowserContext,
)
-from playwright._impl._errors import Error as PlaywrightError
-from ._validators import validate_fetch as _validate
+from ._validators import validate_fetch as _validate, CamoufoxConfig
from ._base import SyncSession, AsyncSession, StealthySessionMixin
from scrapling.core.utils import log
from scrapling.core._types import (
@@ -184,61 +181,21 @@ class StealthySession(StealthySessionMixin, SyncSession):
if self.cookies: # pragma: no cover
self.context.add_cookies(self.cookies)
- def __enter__(self): # pragma: no cover
- self.__create__()
- return self
-
- def __exit__(self, exc_type, exc_val, exc_tb):
- self.close()
-
- def close(self): # pragma: no cover
- """Close all resources"""
- if self._closed: # pragma: no cover
- return
-
- if self.context:
- self.context.close()
- self.context = None
-
- if self.playwright:
- self.playwright.stop()
- self.playwright = None
-
- self._closed = True
-
- @staticmethod
- def _get_page_content(page: Page) -> str:
- """
- A workaround for the Playwright issue with `page.content()` on Windows. Ref.: https://github.com/microsoft/playwright/issues/16108
- :param page: The page to extract content from.
- :return:
- """
- while True:
- try:
- return page.content() or ""
- except PlaywrightError:
- page.wait_for_timeout(1000)
- continue
- return "" # pyright: ignore
-
def _solve_cloudflare(self, page: Page) -> None: # pragma: no cover
"""Solve the cloudflare challenge displayed on the playwright page passed
:param page: The targeted page
:return:
"""
- try:
- page.wait_for_load_state("networkidle", timeout=5000)
- except PlaywrightError:
- pass
- challenge_type = self._detect_cloudflare(self._get_page_content(page))
+ self._wait_for_networkidle(page, timeout=5000)
+ challenge_type = self._detect_cloudflare(ResponseFactory._get_page_content(page))
if not challenge_type:
log.error("No Cloudflare challenge found.")
return
else:
log.info(f'The turnstile version discovered is "{challenge_type}"')
if challenge_type == "non-interactive":
- while "
Just a moment..." in (self._get_page_content(page)):
+ while "Just a moment..." in (ResponseFactory._get_page_content(page)):
log.info("Waiting for Cloudflare wait page to disappear.")
page.wait_for_timeout(1000)
page.wait_for_load_state()
@@ -249,15 +206,14 @@ class StealthySession(StealthySessionMixin, SyncSession):
box_selector = "#cf_turnstile div, #cf-turnstile div, .turnstile>div>div"
if challenge_type != "embedded":
box_selector = ".main-content p+div>div>div"
- while "Verifying you are human." in self._get_page_content(page):
+ while "Verifying you are human." in ResponseFactory._get_page_content(page):
# Waiting for the verify spinner to disappear, checking every 1s if it disappeared
page.wait_for_timeout(500)
outer_box = {}
iframe = page.frame(url=__CF_PATTERN__)
if iframe is not None:
- iframe.wait_for_load_state(state="domcontentloaded")
- iframe.wait_for_load_state("networkidle")
+ self._wait_for_page_stability(iframe, True, True)
if challenge_type != "embedded":
while not iframe.frame_element().is_visible():
@@ -273,16 +229,20 @@ class StealthySession(StealthySessionMixin, SyncSession):
# Move the mouse to the center of the window, then press and hold the left mouse button
page.mouse.click(captcha_x, captcha_y, delay=60, button="left")
- page.wait_for_load_state("networkidle")
+ self._wait_for_networkidle(page)
if iframe is not None:
- # Wait for the frame to be removed from the page
+ # Wait for the frame to be removed from the page (with 30s timeout = 300 iterations * 100 ms)
+ attempts = 0
while iframe in page.frames:
+ if attempts >= 300:
+ log.info("Cloudflare iframe didn't disappear after 30s, continuing...")
+ break
page.wait_for_timeout(100)
+ attempts += 1
if challenge_type != "embedded":
page.locator(box_selector).last.wait_for(state="detached")
page.locator(".zone-name-title").wait_for(state="hidden")
- page.wait_for_load_state(state="load")
- page.wait_for_load_state(state="domcontentloaded")
+ self._wait_for_page_stability(page, True, False)
log.info("Cloudflare captcha is solved")
return
@@ -337,38 +297,26 @@ class StealthySession(StealthySessionMixin, SyncSession):
("solve_cloudflare", solve_cloudflare, self.solve_cloudflare),
("selector_config", selector_config, self.selector_config),
],
+ CamoufoxConfig,
_UNSET,
)
if self._closed: # pragma: no cover
raise RuntimeError("Context manager has been closed")
- final_response = None
referer = (
generate_convincing_referer(url) if (params.google_search and "referer" not in self._headers_keys) else None
)
- def handle_response(finished_response: SyncPlaywrightResponse):
- nonlocal final_response
- if (
- finished_response.request.resource_type == "document"
- and finished_response.request.is_navigation_request()
- and finished_response.request.frame == page_info.page.main_frame
- ):
- final_response = finished_response
-
page_info = self._get_page(params.timeout, params.extra_headers, params.disable_resources)
- page_info.mark_busy(url=url)
+ final_response = [None]
+ handle_response = self._create_response_handler(page_info, final_response)
try: # pragma: no cover
# Navigate to URL and wait for a specified state
page_info.page.on("response", handle_response)
first_response = page_info.page.goto(url, referer=referer)
- if params.load_dom:
- page_info.page.wait_for_load_state(state="domcontentloaded")
-
- if params.network_idle:
- page_info.page.wait_for_load_state("networkidle")
+ self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
if not first_response:
raise RuntimeError(f"Failed to get response for {url}")
@@ -376,11 +324,7 @@ class StealthySession(StealthySessionMixin, SyncSession):
if params.solve_cloudflare:
self._solve_cloudflare(page_info.page)
# Make sure the page is fully loaded after the captcha
- page_info.page.wait_for_load_state(state="load")
- if params.load_dom:
- page_info.page.wait_for_load_state(state="domcontentloaded")
- if params.network_idle:
- page_info.page.wait_for_load_state("networkidle")
+ self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
if params.page_action:
try:
@@ -393,17 +337,13 @@ class StealthySession(StealthySessionMixin, SyncSession):
waiter: Locator = page_info.page.locator(params.wait_selector)
waiter.first.wait_for(state=params.wait_selector_state)
# Wait again after waiting for the selector, helpful with protections like Cloudflare
- page_info.page.wait_for_load_state(state="load")
- if params.load_dom:
- page_info.page.wait_for_load_state(state="domcontentloaded")
- if params.network_idle:
- page_info.page.wait_for_load_state("networkidle")
+ self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
except Exception as e:
log.error(f"Error waiting for selector {params.wait_selector}: {e}")
page_info.page.wait_for_timeout(params.wait)
response = ResponseFactory.from_playwright_response(
- page_info.page, first_response, final_response, params.selector_config
+ page_info.page, first_response, final_response[0], params.selector_config, bool(params.page_action)
)
# Close the page to free up resources
@@ -528,61 +468,21 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
if self.cookies:
await self.context.add_cookies(self.cookies) # pyright: ignore [reportArgumentType]
- async def __aenter__(self):
- await self.__create__()
- return self
-
- async def __aexit__(self, exc_type, exc_val, exc_tb):
- await self.close()
-
- async def close(self):
- """Close all resources"""
- if self._closed: # pragma: no cover
- return
-
- if self.context:
- await self.context.close()
- self.context = None # pyright: ignore
-
- if self.playwright:
- await self.playwright.stop()
- self.playwright = None # pyright: ignore
-
- self._closed = True
-
- @staticmethod
- async def _get_page_content(page: async_Page) -> str:
- """
- A workaround for the Playwright issue with `page.content()` on Windows. Ref.: https://github.com/microsoft/playwright/issues/16108
- :param page: The page to extract content from.
- :return:
- """
- while True:
- try:
- return (await page.content()) or ""
- except PlaywrightError:
- await page.wait_for_timeout(1000)
- continue
- return "" # pyright: ignore
-
- async def _solve_cloudflare(self, page: async_Page):
+ async def _solve_cloudflare(self, page: async_Page): # pragma: no cover
"""Solve the cloudflare challenge displayed on the playwright page passed. The async version
:param page: The async targeted page
:return:
"""
- try:
- await page.wait_for_load_state("networkidle", timeout=5000)
- except PlaywrightError:
- pass
- challenge_type = self._detect_cloudflare(await self._get_page_content(page))
+ await self._wait_for_networkidle(page, timeout=5000)
+ challenge_type = self._detect_cloudflare(await ResponseFactory._get_async_page_content(page))
if not challenge_type:
log.error("No Cloudflare challenge found.")
return
else:
log.info(f'The turnstile version discovered is "{challenge_type}"')
if challenge_type == "non-interactive": # pragma: no cover
- while "Just a moment..." in (await self._get_page_content(page)):
+ while "Just a moment..." in (await ResponseFactory._get_async_page_content(page)):
log.info("Waiting for Cloudflare wait page to disappear.")
await page.wait_for_timeout(1000)
await page.wait_for_load_state()
@@ -593,15 +493,14 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
box_selector = "#cf_turnstile div, #cf-turnstile div, .turnstile>div>div"
if challenge_type != "embedded":
box_selector = ".main-content p+div>div>div"
- while "Verifying you are human." in (await self._get_page_content(page)):
+ while "Verifying you are human." in (await ResponseFactory._get_async_page_content(page)):
# Waiting for the verify spinner to disappear, checking every 1s if it disappeared
await page.wait_for_timeout(500)
outer_box = {}
iframe = page.frame(url=__CF_PATTERN__)
if iframe is not None:
- await iframe.wait_for_load_state(state="domcontentloaded")
- await iframe.wait_for_load_state("networkidle")
+ await self._wait_for_page_stability(iframe, True, True)
if challenge_type != "embedded":
while not await (await iframe.frame_element()).is_visible():
@@ -617,16 +516,20 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
# Move the mouse to the center of the window, then press and hold the left mouse button
await page.mouse.click(captcha_x, captcha_y, delay=60, button="left")
- await page.wait_for_load_state("networkidle")
+ await self._wait_for_networkidle(page)
if iframe is not None:
- # Wait for the frame to be removed from the page
+ # Wait for the frame to be removed from the page (with 30s timeout = 300 iterations * 100 ms)
+ attempts = 0
while iframe in page.frames:
+ if attempts >= 300:
+ log.info("Cloudflare iframe didn't disappear after 30s, continuing...")
+ break
await page.wait_for_timeout(100)
+ attempts += 1
if challenge_type != "embedded":
await page.locator(box_selector).wait_for(state="detached")
await page.locator(".zone-name-title").wait_for(state="hidden")
- await page.wait_for_load_state(state="load")
- await page.wait_for_load_state(state="domcontentloaded")
+ await self._wait_for_page_stability(page, True, False)
log.info("Cloudflare captcha is solved")
return
@@ -681,28 +584,20 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
("solve_cloudflare", solve_cloudflare, self.solve_cloudflare),
("selector_config", selector_config, self.selector_config),
],
+ CamoufoxConfig,
_UNSET,
)
if self._closed: # pragma: no cover
raise RuntimeError("Context manager has been closed")
- final_response = None
referer = (
generate_convincing_referer(url) if (params.google_search and "referer" not in self._headers_keys) else None
)
- async def handle_response(finished_response: AsyncPlaywrightResponse):
- nonlocal final_response
- if (
- finished_response.request.resource_type == "document"
- and finished_response.request.is_navigation_request()
- and finished_response.request.frame == page_info.page.main_frame
- ):
- final_response = finished_response
-
page_info = await self._get_page(params.timeout, params.extra_headers, params.disable_resources)
- page_info.mark_busy(url=url)
+ final_response = [None]
+ handle_response = self._create_response_handler(page_info, final_response)
if TYPE_CHECKING:
if not isinstance(page_info.page, async_Page):
@@ -712,11 +607,7 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
# Navigate to URL and wait for a specified state
page_info.page.on("response", handle_response)
first_response = await page_info.page.goto(url, referer=referer)
- if params.load_dom:
- await page_info.page.wait_for_load_state(state="domcontentloaded")
-
- if params.network_idle:
- await page_info.page.wait_for_load_state("networkidle")
+ await self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
if not first_response:
raise RuntimeError(f"Failed to get response for {url}")
@@ -724,11 +615,7 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
if params.solve_cloudflare:
await self._solve_cloudflare(page_info.page)
# Make sure the page is fully loaded after the captcha
- await page_info.page.wait_for_load_state(state="load")
- if params.load_dom:
- await page_info.page.wait_for_load_state(state="domcontentloaded")
- if params.network_idle:
- await page_info.page.wait_for_load_state("networkidle")
+ await self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
if params.page_action:
try:
@@ -741,11 +628,7 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
waiter: AsyncLocator = page_info.page.locator(params.wait_selector)
await waiter.first.wait_for(state=params.wait_selector_state)
# Wait again after waiting for the selector, helpful with protections like Cloudflare
- await page_info.page.wait_for_load_state(state="load")
- if params.load_dom:
- await page_info.page.wait_for_load_state(state="domcontentloaded")
- if params.network_idle:
- await page_info.page.wait_for_load_state("networkidle")
+ await self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
except Exception as e:
log.error(f"Error waiting for selector {params.wait_selector}: {e}")
@@ -753,7 +636,7 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
# Create response object
response = await ResponseFactory.from_async_playwright_response(
- page_info.page, first_response, final_response, params.selector_config
+ page_info.page, first_response, final_response[0], params.selector_config, bool(params.page_action)
)
# Close the page to free up resources
diff --git a/scrapling/engines/_browsers/_config_tools.py b/scrapling/engines/_browsers/_config_tools.py
index 322d7ad..57e2f17 100644
--- a/scrapling/engines/_browsers/_config_tools.py
+++ b/scrapling/engines/_browsers/_config_tools.py
@@ -70,12 +70,17 @@ def _launch_kwargs(
stealth,
hide_canvas,
disable_webgl,
+ extra_flags: Tuple,
) -> Tuple:
"""Creates the arguments we will use while launching playwright's browser"""
+ base_args = DEFAULT_FLAGS
+ if extra_flags:
+ base_args = base_args + extra_flags
+
launch_kwargs = {
"locale": locale,
"headless": headless,
- "args": DEFAULT_FLAGS,
+ "args": base_args,
"color_scheme": "dark", # Bypasses the 'prefersLightColor' check in creepjs
"proxy": proxy or tuple(),
"device_scale_factor": 2,
@@ -85,9 +90,10 @@ def _launch_kwargs(
"user_agent": useragent or __default_useragent__,
}
if stealth:
+ stealth_args = base_args + _set_flags(hide_canvas, disable_webgl)
launch_kwargs.update(
{
- "args": DEFAULT_FLAGS + _set_flags(hide_canvas, disable_webgl),
+ "args": stealth_args,
"chromium_sandbox": True,
"is_mobile": False,
"has_touch": False,
diff --git a/scrapling/engines/_browsers/_controllers.py b/scrapling/engines/_browsers/_controllers.py
index ca8b45e..bcfd0ac 100644
--- a/scrapling/engines/_browsers/_controllers.py
+++ b/scrapling/engines/_browsers/_controllers.py
@@ -1,23 +1,20 @@
from playwright.sync_api import (
- Response as SyncPlaywrightResponse,
- sync_playwright,
- Playwright,
Locator,
+ Playwright,
+ sync_playwright,
)
from playwright.async_api import (
async_playwright,
- Response as AsyncPlaywrightResponse,
- BrowserContext as AsyncBrowserContext,
- Playwright as AsyncPlaywright,
Locator as AsyncLocator,
- Page as async_Page,
+ Playwright as AsyncPlaywright,
+ BrowserContext as AsyncBrowserContext,
)
from patchright.sync_api import sync_playwright as sync_patchright
from patchright.async_api import async_playwright as async_patchright
from scrapling.core.utils import log
from ._base import SyncSession, AsyncSession, DynamicSessionMixin
-from ._validators import validate_fetch as _validate
+from ._validators import validate_fetch as _validate, PlaywrightConfig
from scrapling.core._types import (
Any,
Dict,
@@ -98,6 +95,7 @@ class DynamicSession(DynamicSessionMixin, SyncSession):
load_dom: bool = True,
wait_selector_state: SelectorWaitStates = "attached",
user_data_dir: str = "",
+ extra_flags: Optional[List[str]] = None,
selector_config: Optional[Dict] = None,
additional_args: Optional[Dict] = None,
):
@@ -127,6 +125,7 @@ class DynamicSession(DynamicSessionMixin, SyncSession):
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
:param user_data_dir: Path to a User Data Directory, which stores browser session data like cookies and local storage. The default is to create a temporary directory.
+ :param extra_flags: A list of additional browser flags to pass to the browser on launch.
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings.
"""
@@ -152,6 +151,7 @@ class DynamicSession(DynamicSessionMixin, SyncSession):
extra_headers=extra_headers,
wait_selector=wait_selector,
disable_webgl=disable_webgl,
+ extra_flags=extra_flags,
selector_config=selector_config,
additional_args=additional_args,
disable_resources=disable_resources,
@@ -178,28 +178,6 @@ class DynamicSession(DynamicSessionMixin, SyncSession):
if self.cookies: # pragma: no cover
self.context.add_cookies(self.cookies)
- def __enter__(self):
- self.__create__()
- return self
-
- def __exit__(self, exc_type, exc_val, exc_tb):
- self.close()
-
- def close(self): # pragma: no cover
- """Close all resources"""
- if self._closed:
- return
-
- if self.context:
- self.context.close()
- self.context = None
-
- if self.playwright:
- self.playwright.stop()
- self.playwright = None # pyright: ignore
-
- self._closed = True
-
def fetch(
self,
url: str,
@@ -247,38 +225,26 @@ class DynamicSession(DynamicSessionMixin, SyncSession):
("load_dom", load_dom, self.load_dom),
("selector_config", selector_config, self.selector_config),
],
+ PlaywrightConfig,
_UNSET,
)
if self._closed: # pragma: no cover
raise RuntimeError("Context manager has been closed")
- final_response = None
referer = (
generate_convincing_referer(url) if (params.google_search and "referer" not in self._headers_keys) else None
)
- def handle_response(finished_response: SyncPlaywrightResponse):
- nonlocal final_response
- if (
- finished_response.request.resource_type == "document"
- and finished_response.request.is_navigation_request()
- and finished_response.request.frame == page_info.page.main_frame
- ):
- final_response = finished_response
-
page_info = self._get_page(params.timeout, params.extra_headers, params.disable_resources)
- page_info.mark_busy(url=url)
+ final_response = [None]
+ handle_response = self._create_response_handler(page_info, final_response)
try: # pragma: no cover
# Navigate to URL and wait for a specified state
page_info.page.on("response", handle_response)
first_response = page_info.page.goto(url, referer=referer)
- if params.load_dom:
- page_info.page.wait_for_load_state(state="domcontentloaded")
-
- if params.network_idle:
- page_info.page.wait_for_load_state("networkidle")
+ self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
if not first_response:
raise RuntimeError(f"Failed to get response for {url}")
@@ -294,11 +260,7 @@ class DynamicSession(DynamicSessionMixin, SyncSession):
waiter: Locator = page_info.page.locator(params.wait_selector)
waiter.first.wait_for(state=params.wait_selector_state)
# Wait again after waiting for the selector, helpful with protections like Cloudflare
- page_info.page.wait_for_load_state(state="load")
- if params.load_dom:
- page_info.page.wait_for_load_state(state="domcontentloaded")
- if params.network_idle:
- page_info.page.wait_for_load_state("networkidle")
+ self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
except Exception as e: # pragma: no cover
log.error(f"Error waiting for selector {params.wait_selector}: {e}")
@@ -306,7 +268,7 @@ class DynamicSession(DynamicSessionMixin, SyncSession):
# Create response object
response = ResponseFactory.from_playwright_response(
- page_info.page, first_response, final_response, params.selector_config
+ page_info.page, first_response, final_response[0], params.selector_config, bool(params.page_action)
)
# Close the page to free up resources
@@ -348,6 +310,7 @@ class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
load_dom: bool = True,
wait_selector_state: SelectorWaitStates = "attached",
user_data_dir: str = "",
+ extra_flags: Optional[List[str]] = None,
selector_config: Optional[Dict] = None,
additional_args: Optional[Dict] = None,
):
@@ -378,6 +341,7 @@ class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
:param max_pages: The maximum number of tabs to be opened at the same time. It will be used in rotation through a PagePool.
:param user_data_dir: Path to a User Data Directory, which stores browser session data like cookies and local storage. The default is to create a temporary directory.
+ :param extra_flags: A list of additional browser flags to pass to the browser on launch.
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings.
"""
@@ -404,6 +368,7 @@ class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
extra_headers=extra_headers,
wait_selector=wait_selector,
disable_webgl=disable_webgl,
+ extra_flags=extra_flags,
selector_config=selector_config,
additional_args=additional_args,
disable_resources=disable_resources,
@@ -431,28 +396,6 @@ class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
if self.cookies:
await self.context.add_cookies(self.cookies) # pyright: ignore
- async def __aenter__(self):
- await self.__create__()
- return self
-
- async def __aexit__(self, exc_type, exc_val, exc_tb):
- await self.close()
-
- async def close(self):
- """Close all resources"""
- if self._closed: # pragma: no cover
- return
-
- if self.context:
- await self.context.close()
- self.context = None # pyright: ignore
-
- if self.playwright:
- await self.playwright.stop()
- self.playwright = None # pyright: ignore
-
- self._closed = True
-
async def fetch(
self,
url: str,
@@ -500,30 +443,24 @@ class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
("load_dom", load_dom, self.load_dom),
("selector_config", selector_config, self.selector_config),
],
+ PlaywrightConfig,
_UNSET,
)
if self._closed: # pragma: no cover
raise RuntimeError("Context manager has been closed")
- final_response = None
referer = (
generate_convincing_referer(url) if (params.google_search and "referer" not in self._headers_keys) else None
)
- async def handle_response(finished_response: AsyncPlaywrightResponse):
- nonlocal final_response
- if (
- finished_response.request.resource_type == "document"
- and finished_response.request.is_navigation_request()
- and finished_response.request.frame == page_info.page.main_frame
- ):
- final_response = finished_response
-
page_info = await self._get_page(params.timeout, params.extra_headers, params.disable_resources)
- page_info.mark_busy(url=url)
+ final_response = [None]
+ handle_response = self._create_response_handler(page_info, final_response)
if TYPE_CHECKING:
+ from playwright.async_api import Page as async_Page
+
if not isinstance(page_info.page, async_Page):
raise TypeError
@@ -531,11 +468,7 @@ class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
# Navigate to URL and wait for a specified state
page_info.page.on("response", handle_response)
first_response = await page_info.page.goto(url, referer=referer)
- if self.load_dom:
- await page_info.page.wait_for_load_state(state="domcontentloaded")
-
- if params.network_idle:
- await page_info.page.wait_for_load_state("networkidle")
+ await self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
if not first_response:
raise RuntimeError(f"Failed to get response for {url}")
@@ -551,11 +484,7 @@ class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
waiter: AsyncLocator = page_info.page.locator(params.wait_selector)
await waiter.first.wait_for(state=params.wait_selector_state)
# Wait again after waiting for the selector, helpful with protections like Cloudflare
- await page_info.page.wait_for_load_state(state="load")
- if self.load_dom:
- await page_info.page.wait_for_load_state(state="domcontentloaded")
- if params.network_idle:
- await page_info.page.wait_for_load_state("networkidle")
+ await self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
except Exception as e:
log.error(f"Error waiting for selector {params.wait_selector}: {e}")
@@ -563,7 +492,7 @@ class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
# Create response object
response = await ResponseFactory.from_async_playwright_response(
- page_info.page, first_response, final_response, params.selector_config
+ page_info.page, first_response, final_response[0], params.selector_config, bool(params.page_action)
)
# Close the page to free up resources
diff --git a/scrapling/engines/_browsers/_validators.py b/scrapling/engines/_browsers/_validators.py
index 0d7d236..edeaa2f 100644
--- a/scrapling/engines/_browsers/_validators.py
+++ b/scrapling/engines/_browsers/_validators.py
@@ -1,7 +1,8 @@
from pathlib import Path
from typing import Annotated
-from dataclasses import dataclass
+from functools import lru_cache
from urllib.parse import urlparse
+from dataclasses import dataclass, fields
from msgspec import Struct, Meta, convert, ValidationError
@@ -19,18 +20,20 @@ from scrapling.engines.toolbelt.navigation import construct_proxy_dict
# Custom validators for msgspec
-def _validate_file_path(value: str):
+@lru_cache(8)
+def _is_invalid_file_path(value: str) -> bool | str:
"""Fast file path validation"""
path = Path(value)
if not path.exists():
- raise ValueError(f"Init script path not found: {value}")
+ return f"Init script path not found: {value}"
if not path.is_file():
- raise ValueError(f"Init script is not a file: {value}")
+ return f"Init script is not a file: {value}"
if not path.is_absolute():
- raise ValueError(f"Init script is not a absolute path: {value}")
+ return f"Init script is not a absolute path: {value}"
+ return False
-def _validate_addon_path(value: str):
+def _validate_addon_path(value: str) -> None:
"""Fast addon path validation"""
path = Path(value)
if not path.exists():
@@ -39,22 +42,16 @@ def _validate_addon_path(value: str):
raise ValueError(f"Addon path must be a directory of the extracted addon: {value}")
-def _validate_cdp_url(cdp_url: str):
+@lru_cache(2)
+def _is_invalid_cdp_url(cdp_url: str) -> bool | str:
"""Fast CDP URL validation"""
- try:
- # Check the scheme
- if not cdp_url.startswith(("ws://", "wss://")):
- raise ValueError("CDP URL must use 'ws://' or 'wss://' scheme")
+ if not cdp_url.startswith(("ws://", "wss://")):
+ return "CDP URL must use 'ws://' or 'wss://' scheme"
- # Validate hostname and port
- if not urlparse(cdp_url).netloc:
- raise ValueError("Invalid hostname for the CDP URL")
-
- except AttributeError as e:
- raise ValueError(f"Malformed CDP URL: {cdp_url}: {str(e)}")
-
- except Exception as e:
- raise ValueError(f"Invalid CDP URL '{cdp_url}': {str(e)}")
+ netloc = urlparse(cdp_url).netloc
+ if not netloc:
+ return "Invalid hostname for the CDP URL"
+ return False
# Type aliases for cleaner annotations
@@ -62,7 +59,7 @@ PagesCount = Annotated[int, Meta(ge=1, le=50)]
Seconds = Annotated[int, float, Meta(ge=0)]
-class PlaywrightConfig(Struct, kw_only=True, frozen=False):
+class PlaywrightConfig(Struct, kw_only=True, frozen=False, weakref=True):
"""Configuration struct for validation"""
max_pages: PagesCount = 1
@@ -88,6 +85,7 @@ class PlaywrightConfig(Struct, kw_only=True, frozen=False):
load_dom: bool = True
wait_selector_state: SelectorWaitStates = "attached"
user_data_dir: str = ""
+ extra_flags: Optional[List[str]] = None
selector_config: Optional[Dict] = {}
additional_args: Optional[Dict] = {}
@@ -98,20 +96,26 @@ class PlaywrightConfig(Struct, kw_only=True, frozen=False):
if self.proxy:
self.proxy = construct_proxy_dict(self.proxy, as_tuple=True)
if self.cdp_url:
- _validate_cdp_url(self.cdp_url)
+ cdp_msg = _is_invalid_cdp_url(self.cdp_url)
+ if cdp_msg:
+ raise ValueError(cdp_msg)
if not self.cookies:
self.cookies = []
+ if not self.extra_flags:
+ self.extra_flags = []
if not self.selector_config:
self.selector_config = {}
if not self.additional_args:
self.additional_args = {}
if self.init_script is not None:
- _validate_file_path(self.init_script)
+ validation_msg = _is_invalid_file_path(self.init_script)
+ if validation_msg:
+ raise ValueError(validation_msg)
-class CamoufoxConfig(Struct, kw_only=True, frozen=False):
+class CamoufoxConfig(Struct, kw_only=True, frozen=False, weakref=True):
"""Configuration struct for validation"""
max_pages: PagesCount = 1
@@ -149,14 +153,16 @@ class CamoufoxConfig(Struct, kw_only=True, frozen=False):
if self.proxy:
self.proxy = construct_proxy_dict(self.proxy, as_tuple=True)
- if self.addons and isinstance(self.addons, list):
+ if self.addons:
for addon in self.addons:
_validate_addon_path(addon)
else:
self.addons = []
if self.init_script is not None:
- _validate_file_path(self.init_script)
+ validation_msg = _is_invalid_file_path(self.init_script)
+ if validation_msg:
+ raise ValueError(validation_msg)
if not self.cookies:
self.cookies = []
@@ -169,27 +175,6 @@ class CamoufoxConfig(Struct, kw_only=True, frozen=False):
self.additional_args = {}
-# Code parts to validate `fetch` in the least possible numbers of lines overall
-class FetchConfig(Struct, kw_only=True):
- """Configuration struct for `fetch` calls validation"""
-
- google_search: bool = True
- timeout: Seconds = 30000
- wait: Seconds = 0
- page_action: Optional[Callable] = None
- extra_headers: Optional[Dict[str, str]] = None
- disable_resources: bool = False
- wait_selector: Optional[str] = None
- wait_selector_state: SelectorWaitStates = "attached"
- network_idle: bool = False
- load_dom: bool = True
- solve_cloudflare: bool = False
- selector_config: Dict = {}
-
- def to_dict(self):
- return {f: getattr(self, f) for f in self.__struct_fields__}
-
-
@dataclass
class _fetch_params:
"""A dataclass of all parameters used by `fetch` calls"""
@@ -208,7 +193,9 @@ class _fetch_params:
selector_config: Dict
-def validate_fetch(params: List[Tuple], sentinel=None) -> _fetch_params:
+def validate_fetch(
+ params: List[Tuple], model: type[PlaywrightConfig] | type[CamoufoxConfig], sentinel=None
+) -> _fetch_params:
result = {}
overrides = {}
@@ -219,16 +206,44 @@ def validate_fetch(params: List[Tuple], sentinel=None) -> _fetch_params:
result[arg] = session_value
if overrides:
- overrides = validate(overrides, FetchConfig).to_dict()
- overrides.update(result)
- return _fetch_params(**overrides)
+ validated_config = validate(overrides, model)
+ # Extract only the fields that _fetch_params needs from validated_config
+ validated_dict = {
+ f.name: getattr(validated_config, f.name)
+ for f in fields(_fetch_params)
+ if hasattr(validated_config, f.name)
+ }
+ # solve_cloudflare defaults to False for models that don't have it (PlaywrightConfig)
+ validated_dict.setdefault("solve_cloudflare", False)
- if not result.get("solve_cloudflare"):
- result["solve_cloudflare"] = False
+ validated_dict.update(result)
+ return _fetch_params(**validated_dict)
+
+ result.setdefault("solve_cloudflare", False)
return _fetch_params(**result)
+# Cache default values for each model to reduce validation overhead
+models_default_values = {}
+
+for _model in (CamoufoxConfig, PlaywrightConfig):
+ _defaults = {}
+ if hasattr(_model, "__struct_defaults__") and hasattr(_model, "__struct_fields__"):
+ for field_name, default_value in zip(_model.__struct_fields__, _model.__struct_defaults__): # type: ignore
+ # Skip factory defaults - these are msgspec._core.Factory instances
+ if type(default_value).__name__ != "Factory":
+ _defaults[field_name] = default_value
+
+ models_default_values[_model.__name__] = _defaults.copy()
+
+
+def _filter_defaults(params: Dict, model: str) -> Dict:
+ """Filter out parameters that match their default values to reduce validation overhead."""
+ defaults = models_default_values[model]
+ return {k: v for k, v in params.items() if k not in defaults or v != defaults[k]}
+
+
@overload
def validate(params: Dict, model: type[PlaywrightConfig]) -> PlaywrightConfig: ...
@@ -237,14 +252,10 @@ def validate(params: Dict, model: type[PlaywrightConfig]) -> PlaywrightConfig: .
def validate(params: Dict, model: type[CamoufoxConfig]) -> CamoufoxConfig: ...
-@overload
-def validate(params: Dict, model: type[FetchConfig]) -> FetchConfig: ...
-
-
-def validate(
- params: Dict, model: type[PlaywrightConfig] | type[CamoufoxConfig] | type[FetchConfig]
-) -> PlaywrightConfig | CamoufoxConfig | FetchConfig:
+def validate(params: Dict, model: type[PlaywrightConfig] | type[CamoufoxConfig]) -> PlaywrightConfig | CamoufoxConfig:
try:
- return convert(params, model)
+ # Filter out params with the default values (no need to validate them) to speed up validation
+ filtered = _filter_defaults(params, model.__name__)
+ return convert(filtered, model)
except ValidationError as e:
raise TypeError(f"Invalid argument type: {e}") from e
diff --git a/scrapling/engines/toolbelt/convertor.py b/scrapling/engines/toolbelt/convertor.py
index b66b518..5bb4a49 100644
--- a/scrapling/engines/toolbelt/convertor.py
+++ b/scrapling/engines/toolbelt/convertor.py
@@ -2,6 +2,7 @@ from functools import lru_cache
from re import compile as re_compile
from curl_cffi.requests import Response as CurlResponse
+from playwright._impl._errors import Error as PlaywrightError
from playwright.sync_api import Page as SyncPage, Response as SyncResponse
from playwright.async_api import Page as AsyncPage, Response as AsyncResponse
@@ -84,6 +85,7 @@ class ResponseFactory:
first_response: SyncResponse,
final_response: Optional[SyncResponse],
parser_arguments: Dict,
+ automated_page: bool = False,
) -> Response:
"""
Transforms a Playwright response into an internal `Response` object, encapsulating
@@ -99,6 +101,7 @@ class ResponseFactory:
:param first_response: An earlier or initial Playwright `Response` object that may serve as a fallback response in the absence of the final one.
:param parser_arguments: A dictionary containing additional arguments needed for parsing or further customization of the returned `Response`. These arguments are dynamically unpacked into
the `Response` object.
+ :param automated_page: If True, it means the `page_action` argument was being used, so the response retrieving method changes to use Playwright's page instead of the final response.
:return: A fully populated `Response` object containing the page's URL, content, status, headers, cookies, and other derived metadata.
:rtype: Response
@@ -114,7 +117,7 @@ class ResponseFactory:
history = cls._process_response_history(first_response, parser_arguments)
try:
- page_content = final_response.text()
+ page_content = final_response.text() if not automated_page else cls._get_page_content(page)
except Exception as e: # pragma: no cover
log.error(f"Error getting page content: {e}")
page_content = ""
@@ -179,6 +182,36 @@ class ResponseFactory:
return history
+ @classmethod
+ def _get_page_content(cls, page: SyncPage) -> str:
+ """
+ A workaround for the Playwright issue with `page.content()` on Windows. Ref.: https://github.com/microsoft/playwright/issues/16108
+ :param page: The page to extract content from.
+ :return:
+ """
+ while True:
+ try:
+ return page.content() or ""
+ except PlaywrightError:
+ page.wait_for_timeout(500)
+ continue
+ return "" # pyright: ignore
+
+ @classmethod
+ async def _get_async_page_content(cls, page: AsyncPage) -> str:
+ """
+ A workaround for the Playwright issue with `page.content()` on Windows. Ref.: https://github.com/microsoft/playwright/issues/16108
+ :param page: The page to extract content from.
+ :return:
+ """
+ while True:
+ try:
+ return (await page.content()) or ""
+ except PlaywrightError:
+ await page.wait_for_timeout(500)
+ continue
+ return "" # pyright: ignore
+
@classmethod
async def from_async_playwright_response(
cls,
@@ -186,6 +219,7 @@ class ResponseFactory:
first_response: AsyncResponse,
final_response: Optional[AsyncResponse],
parser_arguments: Dict,
+ automated_page: bool = False,
) -> Response:
"""
Transforms a Playwright response into an internal `Response` object, encapsulating
@@ -201,6 +235,7 @@ class ResponseFactory:
:param first_response: An earlier or initial Playwright `Response` object that may serve as a fallback response in the absence of the final one.
:param parser_arguments: A dictionary containing additional arguments needed for parsing or further customization of the returned `Response`. These arguments are dynamically unpacked into
the `Response` object.
+ :param automated_page: If True, it means the `page_action` argument was being used, so the response retrieving method changes to use Playwright's page instead of the final response.
:return: A fully populated `Response` object containing the page's URL, content, status, headers, cookies, and other derived metadata.
:rtype: Response
@@ -216,7 +251,7 @@ class ResponseFactory:
history = await cls._async_process_response_history(first_response, parser_arguments)
try:
- page_content = await final_response.text()
+ page_content = await (final_response.text() if not automated_page else cls._get_async_page_content(page))
except Exception as e: # pragma: no cover
log.error(f"Error getting page content in async: {e}")
page_content = ""
diff --git a/scrapling/engines/toolbelt/custom.py b/scrapling/engines/toolbelt/custom.py
index 43ef61c..1f20038 100644
--- a/scrapling/engines/toolbelt/custom.py
+++ b/scrapling/engines/toolbelt/custom.py
@@ -209,15 +209,3 @@ class StatusText:
def get(cls, status_code: int) -> str:
"""Get the phrase for a given HTTP status code."""
return cls._phrases.get(status_code, "Unknown Status Code")
-
-
-def get_variable_name(var: Any) -> Optional[str]:
- """Get the name of a variable using global and local scopes.
- :param var: The variable to find the name for
- :return: The name of the variable if found, None otherwise
- """
- for scope in [globals(), locals()]:
- for name, value in scope.items():
- if value is var:
- return name
- return None
diff --git a/scrapling/engines/toolbelt/fingerprints.py b/scrapling/engines/toolbelt/fingerprints.py
index 7bcad8a..c006ec2 100644
--- a/scrapling/engines/toolbelt/fingerprints.py
+++ b/scrapling/engines/toolbelt/fingerprints.py
@@ -7,8 +7,9 @@ from platform import system as platform_system
from tldextract import extract
from browserforge.headers import Browser, HeaderGenerator
+from browserforge.headers.generator import SUPPORTED_OPERATING_SYSTEMS
-from scrapling.core._types import Dict, Literal
+from scrapling.core._types import Dict, Literal, Tuple
__OS_NAME__ = platform_system()
OSName = Literal["linux", "macos", "windows"]
@@ -29,12 +30,12 @@ def generate_convincing_referer(url: str) -> str:
@lru_cache(1, typed=True)
-def get_os_name() -> OSName | None:
+def get_os_name() -> OSName | Tuple:
"""Get the current OS name in the same format needed for browserforge, if the OS is Unknown, return None so browserforge uses all.
:return: Current OS name or `None` otherwise
"""
- match __OS_NAME__:
+ match __OS_NAME__: # pragma: no cover
case "Linux":
return "linux"
case "Darwin":
@@ -42,7 +43,7 @@ def get_os_name() -> OSName | None:
case "Windows":
return "windows"
case _:
- return None
+ return SUPPORTED_OPERATING_SYSTEMS
def generate_headers(browser_mode: bool = False) -> Dict:
@@ -63,10 +64,7 @@ def generate_headers(browser_mode: bool = False) -> Dict:
Browser(name="edge", min_version=130),
]
)
- if os_name:
- return HeaderGenerator(browser=browsers, os=os_name, device="desktop").generate()
- else:
- return HeaderGenerator(browser=browsers, device="desktop").generate()
+ return HeaderGenerator(browser=browsers, os=os_name, device="desktop").generate()
__default_useragent__ = generate_headers(browser_mode=False).get("User-Agent")
diff --git a/scrapling/fetchers/chrome.py b/scrapling/fetchers/chrome.py
index 9cc980d..0c2ab84 100644
--- a/scrapling/fetchers/chrome.py
+++ b/scrapling/fetchers/chrome.py
@@ -50,6 +50,7 @@ class DynamicFetcher(BaseFetcher):
network_idle: bool = False,
load_dom: bool = True,
wait_selector_state: SelectorWaitStates = "attached",
+ extra_flags: Optional[List[str]] = None,
additional_args: Optional[Dict] = None,
custom_config: Optional[Dict] = None,
) -> Response:
@@ -79,6 +80,7 @@ class DynamicFetcher(BaseFetcher):
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
+ :param extra_flags: A list of additional browser flags to pass to the browser on launch.
:param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values.
:param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings.
:return: A `Response` object.
@@ -108,6 +110,7 @@ class DynamicFetcher(BaseFetcher):
extra_headers=extra_headers,
wait_selector=wait_selector,
disable_webgl=disable_webgl,
+ extra_flags=extra_flags,
additional_args=additional_args,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
@@ -140,6 +143,7 @@ class DynamicFetcher(BaseFetcher):
network_idle: bool = False,
load_dom: bool = True,
wait_selector_state: SelectorWaitStates = "attached",
+ extra_flags: Optional[List[str]] = None,
additional_args: Optional[Dict] = None,
custom_config: Optional[Dict] = None,
) -> Response:
@@ -169,6 +173,7 @@ class DynamicFetcher(BaseFetcher):
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
+ :param extra_flags: A list of additional browser flags to pass to the browser on launch.
:param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values.
:param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings.
:return: A `Response` object.
@@ -199,6 +204,7 @@ class DynamicFetcher(BaseFetcher):
extra_headers=extra_headers,
wait_selector=wait_selector,
disable_webgl=disable_webgl,
+ extra_flags=extra_flags,
additional_args=additional_args,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
diff --git a/setup.cfg b/setup.cfg
index cd31254..fddbe8f 100644
--- a/setup.cfg
+++ b/setup.cfg
@@ -1,6 +1,6 @@
[metadata]
name = scrapling
-version = 0.3.7
+version = 0.3.8
author = Karim Shoair
author_email = karim.shoair@pm.me
description = Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!