From 5e8275c3eb5c18e4688d093f60918e610b484b1a Mon Sep 17 00:00:00 2001 From: Karim shoair Date: Fri, 15 Nov 2024 16:36:32 +0200 Subject: [PATCH] Adding the option to randomize the OS fingerprints with the StealthyFetcher --- README.md | 3 ++- scrapling/engines/camo.py | 20 +++++++++++--------- scrapling/fetchers.py | 3 +++ 3 files changed, 16 insertions(+), 10 deletions(-) diff --git a/README.md b/README.md index d19981e..14b386c 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ Scrapling is a high-performance, intelligent web scraping library for Python tha ```python >> from scrapling import Fetcher, StealthyFetcher, PlayWrightFetcher # Fetch websites' source under the radar! ->> page = StealthyFetcher().fetch('https://example.com', headless=True, disable_resources=True) +>> page = StealthyFetcher().fetch('https://example.com', headless=True, network_idle=True) >> print(page.status) 200 >> products = page.css('.product', auto_save=True) # Scrape data that survives website design changes! @@ -258,6 +258,7 @@ True | timeout | The timeout in milliseconds that is used in all operations and waits through the page. The default is 30000. | ✔️ | | wait_selector | Wait for a specific css selector to be in a specific state. | ✔️ | | proxy | The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. | ✔️ | +| os_randomize | If enabled, Scrapling will randomize the OS fingerprints used. The default is Scrapling matching the fingerprints with the current OS. | ✔️ | | wait_selector_state | The state to wait for the selector given with `wait_selector`. _Default state is `attached`._ | ✔️ | diff --git a/scrapling/engines/camo.py b/scrapling/engines/camo.py index 6a26495..19d8726 100644 --- a/scrapling/engines/camo.py +++ b/scrapling/engines/camo.py @@ -20,7 +20,7 @@ class CamoufoxEngine: block_webrtc: Optional[bool] = False, allow_webgl: Optional[bool] = False, network_idle: Optional[bool] = False, humanize: Optional[Union[bool, float]] = True, timeout: Optional[float] = 30000, page_action: Callable = do_nothing, wait_selector: Optional[str] = None, addons: Optional[List[str]] = None, wait_selector_state: str = 'attached', google_search: Optional[bool] = True, extra_headers: Optional[Dict[str, str]] = None, - proxy: Optional[Union[str, Dict[str, str]]] = None, adaptor_arguments: Dict = None + proxy: Optional[Union[str, Dict[str, str]]] = None, os_randomize: Optional[bool] = None, adaptor_arguments: Dict = None ): """An engine that utilizes Camoufox library, check the `StealthyFetcher` class for more documentation. @@ -35,6 +35,7 @@ class CamoufoxEngine: :param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window. :param allow_webgl: Whether to allow WebGL. To prevent leaks, only use this for special cases. :param network_idle: Wait for the page until there are no network connections for at least 500 ms. + :param os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. The default is Scrapling matching the fingerprints with the current OS. :param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30000 :param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again. :param wait_selector: Wait for a specific css selector to be in a specific state. @@ -51,6 +52,7 @@ class CamoufoxEngine: self.allow_webgl = bool(allow_webgl) self.network_idle = bool(network_idle) self.google_search = bool(google_search) + self.os_randomize = bool(os_randomize) self.extra_headers = extra_headers or {} self.proxy = construct_proxy_dict(proxy) self.addons = addons or [] @@ -73,15 +75,15 @@ class CamoufoxEngine: :return: A Response object with `url`, `text`, `content`, `status`, `reason`, `encoding`, `cookies`, `headers`, `request_headers`, and the `adaptor` class for parsing, of course. """ with Camoufox( - headless=self.headless, - block_images=self.block_images, # Careful! it makes some websites doesn't finish loading at all like stackoverflow even in headful - os=get_os_name(), - block_webrtc=self.block_webrtc, - allow_webgl=self.allow_webgl, - addons=self.addons, - humanize=self.humanize, proxy=self.proxy, - i_know_what_im_doing=True, # To turn warnings off with user configurations + addons=self.addons, + headless=self.headless, + humanize=self.humanize, + i_know_what_im_doing=True, # To turn warnings off with the user configurations + allow_webgl=self.allow_webgl, + block_webrtc=self.block_webrtc, + block_images=self.block_images, # Careful! it makes some websites doesn't finish loading at all like stackoverflow even in headful + os=None if self.os_randomize else get_os_name(), ) as browser: page = browser.new_page() page.set_default_navigation_timeout(self.timeout) diff --git a/scrapling/fetchers.py b/scrapling/fetchers.py index b0f1d60..a63edcb 100644 --- a/scrapling/fetchers.py +++ b/scrapling/fetchers.py @@ -73,6 +73,7 @@ class StealthyFetcher(BaseFetcher): block_webrtc: Optional[bool] = False, allow_webgl: Optional[bool] = False, network_idle: Optional[bool] = False, addons: Optional[List[str]] = None, timeout: Optional[float] = 30000, page_action: Callable = do_nothing, wait_selector: Optional[str] = None, humanize: Optional[Union[bool, float]] = True, wait_selector_state: str = 'attached', google_search: Optional[bool] = True, extra_headers: Optional[Dict[str, str]] = None, proxy: Optional[Union[str, Dict[str, str]]] = None, + os_randomize: Optional[bool] = None ) -> Response: """ Opens up a browser and do your request based on your chosen options below. @@ -88,6 +89,7 @@ class StealthyFetcher(BaseFetcher): :param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window. :param allow_webgl: Whether to allow WebGL. To prevent leaks, only use this for special cases. :param network_idle: Wait for the page until there are no network connections for at least 500 ms. + :param os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. The default is Scrapling matching the fingerprints with the current OS. :param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30000 :param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again. :param wait_selector: Wait for a specific css selector to be in a specific state. @@ -108,6 +110,7 @@ class StealthyFetcher(BaseFetcher): network_idle=network_idle, block_images=block_images, block_webrtc=block_webrtc, + os_randomize=os_randomize, wait_selector=wait_selector, google_search=google_search, extra_headers=extra_headers,