From bf72678480a5f3019038c4bb31fde81d7c6c5645 Mon Sep 17 00:00:00 2001 From: Karim shoair Date: Sat, 10 May 2025 19:36:53 +0300 Subject: [PATCH] feat(cookies): The ability to pass cookies to browser fetchers --- scrapling/engines/camo.py | 10 ++++++++++ scrapling/engines/pw.py | 18 +++++++++++++++++- scrapling/fetchers.py | 13 +++++++++++++ 3 files changed, 40 insertions(+), 1 deletion(-) diff --git a/scrapling/engines/camo.py b/scrapling/engines/camo.py index f3ba36b..64b690a 100644 --- a/scrapling/engines/camo.py +++ b/scrapling/engines/camo.py @@ -14,6 +14,7 @@ from scrapling.core._types import ( Optional, SelectorWaitStates, Union, + Iterable, ) from scrapling.core.utils import log from scrapling.engines.toolbelt import ( @@ -45,6 +46,7 @@ class CamoufoxEngine: wait_selector: Optional[str] = None, addons: Optional[List[str]] = None, wait_selector_state: SelectorWaitStates = "attached", + cookies: Optional[Iterable[Dict]] = None, google_search: bool = True, extra_headers: Optional[Dict[str, str]] = None, proxy: Optional[Union[str, Dict[str, str]]] = None, @@ -63,6 +65,7 @@ class CamoufoxEngine: Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. :param block_webrtc: Blocks WebRTC entirely. + :param cookies: Set cookies for the next request. :param addons: List of Firefox addons to use. Must be paths to extracted addons. :param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window. :param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page before returning the response to you. @@ -97,6 +100,7 @@ class CamoufoxEngine: self.additional_arguments = additional_arguments or {} self.proxy = construct_proxy_dict(proxy) self.addons = addons or [] + self.cookies = cookies or [] self.humanize = humanize self.solve_cloudflare = solve_cloudflare self.timeout = check_type_validity(timeout, [int, float], 30_000) @@ -378,6 +382,9 @@ class CamoufoxEngine: with Camoufox(**self._get_camoufox_options()) as browser: context = browser.new_context() + if self.cookies: + context.add_cookies(self.cookies) + page = context.new_page() page.set_default_navigation_timeout(self.timeout) page.set_default_timeout(self.timeout) @@ -480,6 +487,9 @@ class CamoufoxEngine: async with AsyncCamoufox(**self._get_camoufox_options()) as browser: context = await browser.new_context() + if self.cookies: + await context.add_cookies(self.cookies) + page = await context.new_page() page.set_default_navigation_timeout(self.timeout) page.set_default_timeout(self.timeout) diff --git a/scrapling/engines/pw.py b/scrapling/engines/pw.py index 80b033b..d2d488e 100644 --- a/scrapling/engines/pw.py +++ b/scrapling/engines/pw.py @@ -1,6 +1,13 @@ import json -from scrapling.core._types import Callable, Dict, Optional, SelectorWaitStates, Union +from scrapling.core._types import ( + Callable, + Dict, + Optional, + SelectorWaitStates, + Union, + Iterable, +) from scrapling.core.utils import log, lru_cache from scrapling.engines.constants import DEFAULT_STEALTH_FLAGS, NSTBROWSER_DEFAULT_QUERY from scrapling.engines.toolbelt import ( @@ -30,6 +37,7 @@ class PlaywrightEngine: wait_selector: Optional[str] = None, locale: Optional[str] = "en-US", wait_selector_state: SelectorWaitStates = "attached", + cookies: Optional[Iterable[Dict]] = None, stealth: bool = False, real_chrome: bool = False, hide_canvas: bool = False, @@ -49,6 +57,7 @@ class PlaywrightEngine: Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. :param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it. + :param cookies: Set cookies for the next request. :param network_idle: Wait for the page until there are no network connections for at least 500 ms. :param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 :param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. @@ -81,6 +90,7 @@ class PlaywrightEngine: self.proxy = construct_proxy_dict(proxy) self.cdp_url = cdp_url self.useragent = useragent + self.cookies = cookies or [] self.timeout = check_type_validity(timeout, [int, float], 30000) self.wait = check_type_validity(wait, [int, float], 0) if page_action is not None: @@ -337,6 +347,9 @@ class PlaywrightEngine: browser = p.chromium.launch(**self.__launch_kwargs()) context = browser.new_context(**self.__context_kwargs()) + if self.cookies: + context.add_cookies(self.cookies) + page = context.new_page() page.set_default_navigation_timeout(self.timeout) page.set_default_timeout(self.timeout) @@ -449,6 +462,9 @@ class PlaywrightEngine: browser = await p.chromium.launch(**self.__launch_kwargs()) context = await browser.new_context(**self.__context_kwargs()) + if self.cookies: + await context.add_cookies(self.cookies) + page = await context.new_page() page.set_default_navigation_timeout(self.timeout) page.set_default_timeout(self.timeout) diff --git a/scrapling/fetchers.py b/scrapling/fetchers.py index 41e5eed..bbd4b9f 100644 --- a/scrapling/fetchers.py +++ b/scrapling/fetchers.py @@ -6,6 +6,7 @@ from scrapling.core._types import ( Optional, SelectorWaitStates, Union, + Iterable, ) from scrapling.engines import ( CamoufoxEngine, @@ -484,6 +485,7 @@ class StealthyFetcher(BaseFetcher): allow_webgl: bool = True, network_idle: bool = False, addons: Optional[List[str]] = None, + cookies: Optional[Iterable[Dict]] = None, wait: Optional[int] = 0, timeout: Optional[float] = 30000, page_action: Callable = None, @@ -511,6 +513,7 @@ class StealthyFetcher(BaseFetcher): Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. :param block_webrtc: Blocks WebRTC entirely. + :param cookies: Set cookies for the next request. :param addons: List of Firefox addons to use. Must be paths to extracted addons. :param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window. :param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page before returning the response to you. @@ -545,6 +548,7 @@ class StealthyFetcher(BaseFetcher): geoip=geoip, addons=addons, timeout=timeout, + cookies=cookies, headless=headless, humanize=humanize, disable_ads=disable_ads, @@ -573,6 +577,7 @@ class StealthyFetcher(BaseFetcher): block_images: bool = False, disable_resources: bool = False, block_webrtc: bool = False, + cookies: Optional[Iterable[Dict]] = None, allow_webgl: bool = True, network_idle: bool = False, addons: Optional[List[str]] = None, @@ -603,6 +608,7 @@ class StealthyFetcher(BaseFetcher): Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. :param block_webrtc: Blocks WebRTC entirely. + :param cookies: Set cookies for the next request. :param addons: List of Firefox addons to use. Must be paths to extracted addons. :param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window. :param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page before returning the response to you. @@ -637,6 +643,7 @@ class StealthyFetcher(BaseFetcher): geoip=geoip, addons=addons, timeout=timeout, + cookies=cookies, headless=headless, humanize=humanize, disable_ads=disable_ads, @@ -685,6 +692,7 @@ class PlayWrightFetcher(BaseFetcher): network_idle: bool = False, timeout: Optional[float] = 30000, wait: Optional[int] = 0, + cookies: Optional[Iterable[Dict]] = None, page_action: Optional[Callable] = None, wait_selector: Optional[str] = None, wait_selector_state: SelectorWaitStates = "attached", @@ -712,6 +720,7 @@ class PlayWrightFetcher(BaseFetcher): :param network_idle: Wait for the page until there are no network connections for at least 500 ms. :param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 :param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. + :param cookies: Set cookies for the next request. :param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again. :param wait_selector: Wait for a specific CSS selector to be in a specific state. :param locale: Set the locale for the browser if wanted. The default value is `en-US`. @@ -743,6 +752,7 @@ class PlayWrightFetcher(BaseFetcher): timeout=timeout, stealth=stealth, cdp_url=cdp_url, + cookies=cookies, headless=headless, useragent=useragent, real_chrome=real_chrome, @@ -771,6 +781,7 @@ class PlayWrightFetcher(BaseFetcher): network_idle: bool = False, timeout: Optional[float] = 30000, wait: Optional[int] = 0, + cookies: Optional[Iterable[Dict]] = None, page_action: Optional[Callable] = None, wait_selector: Optional[str] = None, wait_selector_state: SelectorWaitStates = "attached", @@ -796,6 +807,7 @@ class PlayWrightFetcher(BaseFetcher): This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. :param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it. :param network_idle: Wait for the page until there are no network connections for at least 500 ms. + :param cookies: Set cookies for the next request. :param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 :param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. :param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again. @@ -829,6 +841,7 @@ class PlayWrightFetcher(BaseFetcher): timeout=timeout, stealth=stealth, cdp_url=cdp_url, + cookies=cookies, headless=headless, useragent=useragent, real_chrome=real_chrome,