From 880b144af0b0491fde3a24fa6fbeaf946179ecb2 Mon Sep 17 00:00:00 2001 From: Karim shoair Date: Tue, 25 Nov 2025 22:03:45 +0200 Subject: [PATCH] refactor(browser fetchers): Make all the type hints dynamic + Faster validation + Also renamed `custom_config` to `selector_config` so it matches the session class. --- scrapling/fetchers/chrome.py | 246 ++++++++++---------------------- scrapling/fetchers/firefox.py | 254 ++++++++++------------------------ 2 files changed, 142 insertions(+), 358 deletions(-) diff --git a/scrapling/fetchers/chrome.py b/scrapling/fetchers/chrome.py index 0c2ab84..44e7a44 100644 --- a/scrapling/fetchers/chrome.py +++ b/scrapling/fetchers/chrome.py @@ -1,10 +1,5 @@ -from scrapling.core._types import ( - Callable, - List, - Dict, - Optional, - SelectorWaitStates, -) +from scrapling.core._types import Unpack +from scrapling.engines._browsers._types import PlaywrightSession from scrapling.engines.toolbelt.custom import BaseFetcher, Response from scrapling.engines._browsers._controllers import DynamicSession, AsyncDynamicSession @@ -26,190 +21,89 @@ class DynamicFetcher(BaseFetcher): """ @classmethod - def fetch( - cls, - url: str, - headless: bool = True, - google_search: bool = True, - hide_canvas: bool = False, - disable_webgl: bool = False, - real_chrome: bool = False, - stealth: bool = False, - wait: int | float = 0, - page_action: Optional[Callable] = None, - proxy: Optional[str | Dict[str, str]] = None, - locale: str = "en-US", - extra_headers: Optional[Dict[str, str]] = None, - useragent: Optional[str] = None, - cdp_url: Optional[str] = None, - timeout: int | float = 30000, - disable_resources: bool = False, - wait_selector: Optional[str] = None, - init_script: Optional[str] = None, - cookies: Optional[List[Dict]] = None, - network_idle: bool = False, - load_dom: bool = True, - wait_selector_state: SelectorWaitStates = "attached", - extra_flags: Optional[List[str]] = None, - additional_args: Optional[Dict] = None, - custom_config: Optional[Dict] = None, - ) -> Response: + def fetch(cls, url: str, **kwargs: Unpack[PlaywrightSession]) -> Response: """Opens up a browser and do your request based on your chosen options below. :param url: Target url. - :param headless: Run the browser in headless/hidden (default), or headful/visible mode. - :param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. - This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. - :param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it. - :param cookies: Set cookies for the next request. - :param network_idle: Wait for the page until there are no network connections for at least 500 ms. - :param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - :param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - :param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - :param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - :param wait_selector: Wait for a specific CSS selector to be in a specific state. - :param init_script: An absolute path to a JavaScript file to be executed on page creation with this request. - :param locale: Set the locale for the browser if wanted. The default value is `en-US`. - :param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - :param stealth: Enables stealth mode, check the documentation to see what stealth mode does currently. - :param real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it. - :param hide_canvas: Add random noise to canvas operations to prevent fingerprinting. - :param disable_webgl: Disables WebGL and WebGL 2.0 support entirely. - :param cdp_url: Instead of launching a new browser instance, connect to this CDP URL to control real browsers through CDP. - :param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. - :param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - :param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. - :param extra_flags: A list of additional browser flags to pass to the browser on launch. - :param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values. - :param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings. + :param kwargs: Browser session configuration options including: + - headless: Run the browser in headless/hidden (default), or headful/visible mode. + - disable_resources: Drop requests of unnecessary resources for a speed boost. + - useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it. + - cookies: Set cookies for the next request. + - network_idle: Wait for the page until there are no network connections for at least 500 ms. + - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. + - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 + - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the Response object. + - page_action: Added for automation. A function that takes the `page` object and does the automation you need. + - wait_selector: Wait for a specific CSS selector to be in a specific state. + - init_script: An absolute path to a JavaScript file to be executed on page creation with this request. + - locale: Set the locale for the browser if wanted. The default value is `en-US`. + - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. + - stealth: Enables stealth mode, check the documentation to see what stealth mode does currently. + - real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it. + - hide_canvas: Add random noise to canvas operations to prevent fingerprinting. + - disable_webgl: Disables WebGL and WebGL 2.0 support entirely. + - cdp_url: Instead of launching a new browser instance, connect to this CDP URL to control real browsers through CDP. + - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. + - extra_headers: A dictionary of extra headers to add to the request. + - proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. + - extra_flags: A list of additional browser flags to pass to the browser on launch. + - selector_config: The arguments that will be passed in the end while creating the final Selector's class. + - additional_args: Additional arguments to be passed to Playwright's context as additional settings. :return: A `Response` object. """ - if not custom_config: - custom_config = {} - elif not isinstance(custom_config, dict): - raise ValueError(f"The custom parser config must be of type dictionary, got {cls.__class__}") + # Get selector_config from kwargs if provided, otherwise use empty dict + selector_config = kwargs.get("selector_config", {}) + if not isinstance(selector_config, dict): + raise TypeError("Argument `selector_config` must be a dictionary.") - with DynamicSession( - wait=wait, - proxy=proxy, - locale=locale, - timeout=timeout, - stealth=stealth, - cdp_url=cdp_url, - cookies=cookies, - headless=headless, - load_dom=load_dom, - useragent=useragent, - real_chrome=real_chrome, - page_action=page_action, - hide_canvas=hide_canvas, - init_script=init_script, - network_idle=network_idle, - google_search=google_search, - extra_headers=extra_headers, - wait_selector=wait_selector, - disable_webgl=disable_webgl, - extra_flags=extra_flags, - additional_args=additional_args, - disable_resources=disable_resources, - wait_selector_state=wait_selector_state, - selector_config={**cls._generate_parser_arguments(), **custom_config}, - ) as session: + # Merge selector_config with class defaults + kwargs["selector_config"] = {**cls._generate_parser_arguments(), **selector_config} + + with DynamicSession(**kwargs) as session: return session.fetch(url) @classmethod - async def async_fetch( - cls, - url: str, - headless: bool = True, - google_search: bool = True, - hide_canvas: bool = False, - disable_webgl: bool = False, - real_chrome: bool = False, - stealth: bool = False, - wait: int | float = 0, - page_action: Optional[Callable] = None, - proxy: Optional[str | Dict[str, str]] = None, - locale: str = "en-US", - extra_headers: Optional[Dict[str, str]] = None, - useragent: Optional[str] = None, - cdp_url: Optional[str] = None, - timeout: int | float = 30000, - disable_resources: bool = False, - wait_selector: Optional[str] = None, - init_script: Optional[str] = None, - cookies: Optional[List[Dict]] = None, - network_idle: bool = False, - load_dom: bool = True, - wait_selector_state: SelectorWaitStates = "attached", - extra_flags: Optional[List[str]] = None, - additional_args: Optional[Dict] = None, - custom_config: Optional[Dict] = None, - ) -> Response: + async def async_fetch(cls, url: str, **kwargs: Unpack[PlaywrightSession]) -> Response: """Opens up a browser and do your request based on your chosen options below. :param url: Target url. - :param headless: Run the browser in headless/hidden (default), or headful/visible mode. - :param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. - This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. - :param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it. - :param cookies: Set cookies for the next request. - :param network_idle: Wait for the page until there are no network connections for at least 500 ms. - :param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - :param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - :param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - :param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - :param wait_selector: Wait for a specific CSS selector to be in a specific state. - :param init_script: An absolute path to a JavaScript file to be executed on page creation with this request. - :param locale: Set the locale for the browser if wanted. The default value is `en-US`. - :param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - :param stealth: Enables stealth mode, check the documentation to see what stealth mode does currently. - :param real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it. - :param hide_canvas: Add random noise to canvas operations to prevent fingerprinting. - :param disable_webgl: Disables WebGL and WebGL 2.0 support entirely. - :param cdp_url: Instead of launching a new browser instance, connect to this CDP URL to control real browsers through CDP. - :param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. - :param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - :param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. - :param extra_flags: A list of additional browser flags to pass to the browser on launch. - :param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values. - :param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings. + :param kwargs: Browser session configuration options including: + - headless: Run the browser in headless/hidden (default), or headful/visible mode. + - disable_resources: Drop requests of unnecessary resources for a speed boost. + - useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it. + - cookies: Set cookies for the next request. + - network_idle: Wait for the page until there are no network connections for at least 500 ms. + - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. + - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 + - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the Response object. + - page_action: Added for automation. A function that takes the `page` object and does the automation you need. + - wait_selector: Wait for a specific CSS selector to be in a specific state. + - init_script: An absolute path to a JavaScript file to be executed on page creation with this request. + - locale: Set the locale for the browser if wanted. The default value is `en-US`. + - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. + - stealth: Enables stealth mode, check the documentation to see what stealth mode does currently. + - real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it. + - hide_canvas: Add random noise to canvas operations to prevent fingerprinting. + - disable_webgl: Disables WebGL and WebGL 2.0 support entirely. + - cdp_url: Instead of launching a new browser instance, connect to this CDP URL to control real browsers through CDP. + - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. + - extra_headers: A dictionary of extra headers to add to the request. + - proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. + - extra_flags: A list of additional browser flags to pass to the browser on launch. + - selector_config: The arguments that will be passed in the end while creating the final Selector's class. + - additional_args: Additional arguments to be passed to Playwright's context as additional settings. :return: A `Response` object. """ - if not custom_config: - custom_config = {} - elif not isinstance(custom_config, dict): - raise ValueError(f"The custom parser config must be of type dictionary, got {cls.__class__}") + # Get selector_config from kwargs if provided, otherwise use empty dict + selector_config = kwargs.get("selector_config", {}) + if not isinstance(selector_config, dict): + raise TypeError("Argument `selector_config` must be a dictionary.") - async with AsyncDynamicSession( - wait=wait, - max_pages=1, - proxy=proxy, - locale=locale, - timeout=timeout, - stealth=stealth, - cdp_url=cdp_url, - cookies=cookies, - headless=headless, - load_dom=load_dom, - useragent=useragent, - real_chrome=real_chrome, - page_action=page_action, - hide_canvas=hide_canvas, - init_script=init_script, - network_idle=network_idle, - google_search=google_search, - extra_headers=extra_headers, - wait_selector=wait_selector, - disable_webgl=disable_webgl, - extra_flags=extra_flags, - additional_args=additional_args, - disable_resources=disable_resources, - wait_selector_state=wait_selector_state, - selector_config={**cls._generate_parser_arguments(), **custom_config}, - ) as session: + # Merge selector_config with class defaults + kwargs["selector_config"] = {**cls._generate_parser_arguments(), **selector_config} + + async with AsyncDynamicSession(**kwargs) as session: return await session.fetch(url) diff --git a/scrapling/fetchers/firefox.py b/scrapling/fetchers/firefox.py index 5986096..825b917 100644 --- a/scrapling/fetchers/firefox.py +++ b/scrapling/fetchers/firefox.py @@ -1,10 +1,5 @@ -from scrapling.core._types import ( - Callable, - Dict, - List, - Optional, - SelectorWaitStates, -) +from scrapling.core._types import Unpack +from scrapling.engines._browsers._types import CamoufoxSession from scrapling.engines.toolbelt.custom import BaseFetcher, Response from scrapling.engines._browsers._camoufox import StealthySession, AsyncStealthySession @@ -17,196 +12,91 @@ class StealthyFetcher(BaseFetcher): """ @classmethod - def fetch( - cls, - url: str, - headless: bool = True, # noqa: F821 - block_images: bool = False, - disable_resources: bool = False, - block_webrtc: bool = False, - allow_webgl: bool = True, - network_idle: bool = False, - load_dom: bool = True, - humanize: bool | float = True, - solve_cloudflare: bool = False, - wait: int | float = 0, - timeout: int | float = 30000, - page_action: Optional[Callable] = None, - wait_selector: Optional[str] = None, - init_script: Optional[str] = None, - addons: Optional[List[str]] = None, - wait_selector_state: SelectorWaitStates = "attached", - cookies: Optional[List[Dict]] = None, - google_search: bool = True, - extra_headers: Optional[Dict[str, str]] = None, - proxy: Optional[str | Dict[str, str]] = None, - os_randomize: bool = False, - disable_ads: bool = False, - geoip: bool = False, - custom_config: Optional[Dict] = None, - additional_args: Optional[Dict] = None, - ) -> Response: + def fetch(cls, url: str, **kwargs: Unpack[CamoufoxSession]) -> Response: """ Opens up a browser and do your request based on your chosen options below. :param url: Target url. - :param headless: Run the browser in headless/hidden (default), or headful/visible mode. - :param block_images: Prevent the loading of images through Firefox preferences. - This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. - :param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. - This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. - :param block_webrtc: Blocks WebRTC entirely. - :param cookies: Set cookies for the next request. - :param addons: List of Firefox addons to use. Must be paths to extracted addons. - :param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window. - :param solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you. - :param allow_webgl: Enabled by default. Disabling WebGL is not recommended as many WAFs now check if WebGL is enabled. - :param network_idle: Wait for the page until there are no network connections for at least 500 ms. - :param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - :param disable_ads: Disabled by default, this installs the `uBlock Origin` addon on the browser if enabled. - :param os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. The default is Scrapling matching the fingerprints with the current OS. - :param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - :param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - :param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - :param wait_selector: Wait for a specific CSS selector to be in a specific state. - :param init_script: An absolute path to a JavaScript file to be executed on page creation with this request. - :param geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, and spoof the WebRTC IP address. - It will also calculate and spoof the browser's language based on the distribution of language speakers in the target region. - :param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - :param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. - :param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - :param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. - :param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values. - :param additional_args: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings. + :param kwargs: Browser session configuration options including: + - headless: Run the browser in headless/hidden (default), or headful/visible mode. + - block_images: Prevent the loading of images through Firefox preferences. + - disable_resources: Drop requests of unnecessary resources for a speed boost. + - block_webrtc: Blocks WebRTC entirely. + - allow_webgl: Enabled by default. Disabling WebGL is not recommended as many WAFs now check if WebGL is enabled. + - network_idle: Wait for the page until there are no network connections for at least 500 ms. + - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. + - humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. + - solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you. + - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the Response object. + - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 + - page_action: Added for automation. A function that takes the `page` object and does the automation you need. + - wait_selector: Wait for a specific CSS selector to be in a specific state. + - init_script: An absolute path to a JavaScript file to be executed on page creation with this request. + - addons: List of Firefox addons to use. Must be paths to extracted addons. + - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. + - cookies: Set cookies for the next request. + - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. + - extra_headers: A dictionary of extra headers to add to the request. + - proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. + - os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. + - disable_ads: Disabled by default, this installs the `uBlock Origin` addon on the browser if enabled. + - geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, and spoof the WebRTC IP address. + - selector_config: The arguments that will be passed in the end while creating the final Selector's class. + - additional_args: Additional arguments to be passed to Camoufox as additional settings. :return: A `Response` object. """ - if not custom_config: - custom_config = {} + # Get selector_config from kwargs if provided, otherwise use empty dict + selector_config = kwargs.get("selector_config", {}) + if not isinstance(selector_config, dict): + raise TypeError("Argument `selector_config` must be a dictionary.") - with StealthySession( - wait=wait, - proxy=proxy, - geoip=geoip, - addons=addons, - timeout=timeout, - cookies=cookies, - headless=headless, - humanize=humanize, - load_dom=load_dom, - disable_ads=disable_ads, - allow_webgl=allow_webgl, - page_action=page_action, - init_script=init_script, - network_idle=network_idle, - block_images=block_images, - block_webrtc=block_webrtc, - os_randomize=os_randomize, - wait_selector=wait_selector, - google_search=google_search, - extra_headers=extra_headers, - solve_cloudflare=solve_cloudflare, - disable_resources=disable_resources, - wait_selector_state=wait_selector_state, - selector_config={**cls._generate_parser_arguments(), **custom_config}, - additional_args=additional_args or {}, - ) as engine: + # Merge selector_config with class defaults + kwargs["selector_config"] = {**cls._generate_parser_arguments(), **selector_config} + + with StealthySession(**kwargs) as engine: return engine.fetch(url) @classmethod - async def async_fetch( - cls, - url: str, - headless: bool = True, # noqa: F821 - block_images: bool = False, - disable_resources: bool = False, - block_webrtc: bool = False, - allow_webgl: bool = True, - network_idle: bool = False, - load_dom: bool = True, - humanize: bool | float = True, - solve_cloudflare: bool = False, - wait: int | float = 0, - timeout: int | float = 30000, - page_action: Optional[Callable] = None, - wait_selector: Optional[str] = None, - init_script: Optional[str] = None, - addons: Optional[List[str]] = None, - wait_selector_state: SelectorWaitStates = "attached", - cookies: Optional[List[Dict]] = None, - google_search: bool = True, - extra_headers: Optional[Dict[str, str]] = None, - proxy: Optional[str | Dict[str, str]] = None, - os_randomize: bool = False, - disable_ads: bool = False, - geoip: bool = False, - custom_config: Optional[Dict] = None, - additional_args: Optional[Dict] = None, - ) -> Response: + async def async_fetch(cls, url: str, **kwargs: Unpack[CamoufoxSession]) -> Response: """ Opens up a browser and do your request based on your chosen options below. :param url: Target url. - :param headless: Run the browser in headless/hidden (default), or headful/visible mode. - :param block_images: Prevent the loading of images through Firefox preferences. - This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. - :param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. - This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. - :param block_webrtc: Blocks WebRTC entirely. - :param cookies: Set cookies for the next request. - :param addons: List of Firefox addons to use. Must be paths to extracted addons. - :param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window. - :param solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you. - :param allow_webgl: Enabled by default. Disabling WebGL is not recommended as many WAFs now check if WebGL is enabled. - :param network_idle: Wait for the page until there are no network connections for at least 500 ms. - :param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - :param disable_ads: Disabled by default, this installs the `uBlock Origin` addon on the browser if enabled. - :param os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. The default is Scrapling matching the fingerprints with the current OS. - :param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - :param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - :param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - :param wait_selector: Wait for a specific CSS selector to be in a specific state. - :param init_script: An absolute path to a JavaScript file to be executed on page creation with this request. - :param geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, and spoof the WebRTC IP address. - It will also calculate and spoof the browser's language based on the distribution of language speakers in the target region. - :param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - :param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. - :param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - :param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. - :param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values. - :param additional_args: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings. + :param kwargs: Browser session configuration options including: + - headless: Run the browser in headless/hidden (default), or headful/visible mode. + - block_images: Prevent the loading of images through Firefox preferences. + - disable_resources: Drop requests of unnecessary resources for a speed boost. + - block_webrtc: Blocks WebRTC entirely. + - allow_webgl: Enabled by default. Disabling WebGL is not recommended as many WAFs now check if WebGL is enabled. + - network_idle: Wait for the page until there are no network connections for at least 500 ms. + - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. + - humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. + - solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you. + - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the Response object. + - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 + - page_action: Added for automation. A function that takes the `page` object and does the automation you need. + - wait_selector: Wait for a specific CSS selector to be in a specific state. + - init_script: An absolute path to a JavaScript file to be executed on page creation with this request. + - addons: List of Firefox addons to use. Must be paths to extracted addons. + - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. + - cookies: Set cookies for the next request. + - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. + - extra_headers: A dictionary of extra headers to add to the request. + - proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. + - os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. + - disable_ads: Disabled by default, this installs the `uBlock Origin` addon on the browser if enabled. + - geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, and spoof the WebRTC IP address. + - selector_config: The arguments that will be passed in the end while creating the final Selector's class. + - additional_args: Additional arguments to be passed to Camoufox as additional settings. :return: A `Response` object. """ - if not custom_config: - custom_config = {} + # Get selector_config from kwargs if provided, otherwise use empty dict + selector_config = kwargs.get("selector_config", {}) + if not isinstance(selector_config, dict): + raise TypeError("Argument `selector_config` must be a dictionary.") - async with AsyncStealthySession( - wait=wait, - max_pages=1, - proxy=proxy, - geoip=geoip, - addons=addons, - timeout=timeout, - cookies=cookies, - headless=headless, - humanize=humanize, - load_dom=load_dom, - disable_ads=disable_ads, - allow_webgl=allow_webgl, - page_action=page_action, - init_script=init_script, - network_idle=network_idle, - block_images=block_images, - block_webrtc=block_webrtc, - os_randomize=os_randomize, - wait_selector=wait_selector, - google_search=google_search, - extra_headers=extra_headers, - solve_cloudflare=solve_cloudflare, - disable_resources=disable_resources, - wait_selector_state=wait_selector_state, - selector_config={**cls._generate_parser_arguments(), **custom_config}, - additional_args=additional_args or {}, - ) as engine: + # Merge selector_config with class defaults + kwargs["selector_config"] = {**cls._generate_parser_arguments(), **selector_config} + + async with AsyncStealthySession(**kwargs) as engine: return await engine.fetch(url)