From d33e738c53f4bc460f4c4fe9091e3689e5c9ea3b Mon Sep 17 00:00:00 2001 From: Karim shoair Date: Wed, 26 Mar 2025 05:36:05 +0200 Subject: [PATCH] feat(StealthyFetcher): The ability to pass additional arguments to Camoufox --- scrapling/engines/camo.py | 4 ++++ scrapling/fetchers.py | 8 ++++++-- 2 files changed, 10 insertions(+), 2 deletions(-) diff --git a/scrapling/engines/camo.py b/scrapling/engines/camo.py index 7893526..246c308 100644 --- a/scrapling/engines/camo.py +++ b/scrapling/engines/camo.py @@ -22,6 +22,7 @@ class CamoufoxEngine: proxy: Optional[Union[str, Dict[str, str]]] = None, os_randomize: bool = False, disable_ads: bool = False, geoip: bool = False, adaptor_arguments: Dict = None, + **additional_arguments: Dict ): """An engine that utilizes Camoufox library, check the `StealthyFetcher` class for more documentation. @@ -48,6 +49,7 @@ class CamoufoxEngine: :param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ :param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. :param adaptor_arguments: The arguments that will be passed in the end while creating the final Adaptor's class. + :param additional_arguments: Any Additional arguments will be passed to Camoufox as additional settings and takes higher priority than Scrapling's settings. """ self.headless = headless self.block_images = bool(block_images) @@ -60,6 +62,7 @@ class CamoufoxEngine: self.disable_ads = bool(disable_ads) self.geoip = bool(geoip) self.extra_headers = extra_headers or {} + self.additional_arguments = additional_arguments self.proxy = construct_proxy_dict(proxy) self.addons = addons or [] self.humanize = humanize @@ -92,6 +95,7 @@ class CamoufoxEngine: "block_webrtc": self.block_webrtc, "block_images": self.block_images, # Careful! it makes some websites doesn't finish loading at all like stackoverflow even in headful "os": None if self.os_randomize else get_os_name(), + **self.additional_arguments } def _process_response_history(self, first_response): diff --git a/scrapling/fetchers.py b/scrapling/fetchers.py index 643c845..7728402 100644 --- a/scrapling/fetchers.py +++ b/scrapling/fetchers.py @@ -235,7 +235,7 @@ class StealthyFetcher(BaseFetcher): timeout: Optional[float] = 30000, page_action: Callable = None, wait_selector: Optional[str] = None, humanize: Optional[Union[bool, float]] = True, wait_selector_state: SelectorWaitStates = 'attached', google_search: bool = True, extra_headers: Optional[Dict[str, str]] = None, proxy: Optional[Union[str, Dict[str, str]]] = None, os_randomize: bool = False, disable_ads: bool = False, geoip: bool = False, - custom_config: Dict = None + custom_config: Dict = None, **additional_arguments: Dict ) -> Response: """ Opens up a browser and do your request based on your chosen options below. @@ -264,6 +264,7 @@ class StealthyFetcher(BaseFetcher): :param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ :param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. :param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values. + :param additional_arguments: Any Additional arguments will be passed to Camoufox as additional settings and takes higher priority than Scrapling's settings. :return: A `Response` object that is the same as `Adaptor` object except it has these added attributes: `status`, `reason`, `cookies`, `headers`, and `request_headers` """ if not custom_config: @@ -291,6 +292,7 @@ class StealthyFetcher(BaseFetcher): disable_resources=disable_resources, wait_selector_state=wait_selector_state, adaptor_arguments={**cls._generate_parser_arguments(), **custom_config}, + **additional_arguments ) return engine.fetch(url) @@ -301,7 +303,7 @@ class StealthyFetcher(BaseFetcher): timeout: Optional[float] = 30000, page_action: Callable = None, wait_selector: Optional[str] = None, humanize: Optional[Union[bool, float]] = True, wait_selector_state: SelectorWaitStates = 'attached', google_search: bool = True, extra_headers: Optional[Dict[str, str]] = None, proxy: Optional[Union[str, Dict[str, str]]] = None, os_randomize: bool = False, disable_ads: bool = False, geoip: bool = False, - custom_config: Dict = None + custom_config: Dict = None, **additional_arguments: Dict ) -> Response: """ Opens up a browser and do your request based on your chosen options below. @@ -330,6 +332,7 @@ class StealthyFetcher(BaseFetcher): :param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ :param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. :param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values. + :param additional_arguments: Any Additional arguments will be passed to Camoufox as additional settings and takes higher priority than Scrapling's settings. :return: A `Response` object that is the same as `Adaptor` object except it has these added attributes: `status`, `reason`, `cookies`, `headers`, and `request_headers` """ if not custom_config: @@ -357,6 +360,7 @@ class StealthyFetcher(BaseFetcher): disable_resources=disable_resources, wait_selector_state=wait_selector_state, adaptor_arguments={**cls._generate_parser_arguments(), **custom_config}, + **additional_arguments ) return await engine.async_fetch(url)