feat(cookies): The ability to pass cookies to browser fetchers

This commit is contained in:
Karim shoair
2025-05-10 19:36:53 +03:00
parent c84143129e
commit bf72678480
3 changed files with 40 additions and 1 deletions
+10
View File
@@ -14,6 +14,7 @@ from scrapling.core._types import (
Optional, Optional,
SelectorWaitStates, SelectorWaitStates,
Union, Union,
Iterable,
) )
from scrapling.core.utils import log from scrapling.core.utils import log
from scrapling.engines.toolbelt import ( from scrapling.engines.toolbelt import (
@@ -45,6 +46,7 @@ class CamoufoxEngine:
wait_selector: Optional[str] = None, wait_selector: Optional[str] = None,
addons: Optional[List[str]] = None, addons: Optional[List[str]] = None,
wait_selector_state: SelectorWaitStates = "attached", wait_selector_state: SelectorWaitStates = "attached",
cookies: Optional[Iterable[Dict]] = None,
google_search: bool = True, google_search: bool = True,
extra_headers: Optional[Dict[str, str]] = None, extra_headers: Optional[Dict[str, str]] = None,
proxy: Optional[Union[str, Dict[str, str]]] = None, proxy: Optional[Union[str, Dict[str, str]]] = None,
@@ -63,6 +65,7 @@ class CamoufoxEngine:
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
:param block_webrtc: Blocks WebRTC entirely. :param block_webrtc: Blocks WebRTC entirely.
:param cookies: Set cookies for the next request.
:param addons: List of Firefox addons to use. Must be paths to extracted addons. :param addons: List of Firefox addons to use. Must be paths to extracted addons.
:param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window. :param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window.
:param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page before returning the response to you. :param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page before returning the response to you.
@@ -97,6 +100,7 @@ class CamoufoxEngine:
self.additional_arguments = additional_arguments or {} self.additional_arguments = additional_arguments or {}
self.proxy = construct_proxy_dict(proxy) self.proxy = construct_proxy_dict(proxy)
self.addons = addons or [] self.addons = addons or []
self.cookies = cookies or []
self.humanize = humanize self.humanize = humanize
self.solve_cloudflare = solve_cloudflare self.solve_cloudflare = solve_cloudflare
self.timeout = check_type_validity(timeout, [int, float], 30_000) self.timeout = check_type_validity(timeout, [int, float], 30_000)
@@ -378,6 +382,9 @@ class CamoufoxEngine:
with Camoufox(**self._get_camoufox_options()) as browser: with Camoufox(**self._get_camoufox_options()) as browser:
context = browser.new_context() context = browser.new_context()
if self.cookies:
context.add_cookies(self.cookies)
page = context.new_page() page = context.new_page()
page.set_default_navigation_timeout(self.timeout) page.set_default_navigation_timeout(self.timeout)
page.set_default_timeout(self.timeout) page.set_default_timeout(self.timeout)
@@ -480,6 +487,9 @@ class CamoufoxEngine:
async with AsyncCamoufox(**self._get_camoufox_options()) as browser: async with AsyncCamoufox(**self._get_camoufox_options()) as browser:
context = await browser.new_context() context = await browser.new_context()
if self.cookies:
await context.add_cookies(self.cookies)
page = await context.new_page() page = await context.new_page()
page.set_default_navigation_timeout(self.timeout) page.set_default_navigation_timeout(self.timeout)
page.set_default_timeout(self.timeout) page.set_default_timeout(self.timeout)
+17 -1
View File
@@ -1,6 +1,13 @@
import json import json
from scrapling.core._types import Callable, Dict, Optional, SelectorWaitStates, Union from scrapling.core._types import (
Callable,
Dict,
Optional,
SelectorWaitStates,
Union,
Iterable,
)
from scrapling.core.utils import log, lru_cache from scrapling.core.utils import log, lru_cache
from scrapling.engines.constants import DEFAULT_STEALTH_FLAGS, NSTBROWSER_DEFAULT_QUERY from scrapling.engines.constants import DEFAULT_STEALTH_FLAGS, NSTBROWSER_DEFAULT_QUERY
from scrapling.engines.toolbelt import ( from scrapling.engines.toolbelt import (
@@ -30,6 +37,7 @@ class PlaywrightEngine:
wait_selector: Optional[str] = None, wait_selector: Optional[str] = None,
locale: Optional[str] = "en-US", locale: Optional[str] = "en-US",
wait_selector_state: SelectorWaitStates = "attached", wait_selector_state: SelectorWaitStates = "attached",
cookies: Optional[Iterable[Dict]] = None,
stealth: bool = False, stealth: bool = False,
real_chrome: bool = False, real_chrome: bool = False,
hide_canvas: bool = False, hide_canvas: bool = False,
@@ -49,6 +57,7 @@ class PlaywrightEngine:
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
:param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it. :param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it.
:param cookies: Set cookies for the next request.
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. :param network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 :param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. :param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
@@ -81,6 +90,7 @@ class PlaywrightEngine:
self.proxy = construct_proxy_dict(proxy) self.proxy = construct_proxy_dict(proxy)
self.cdp_url = cdp_url self.cdp_url = cdp_url
self.useragent = useragent self.useragent = useragent
self.cookies = cookies or []
self.timeout = check_type_validity(timeout, [int, float], 30000) self.timeout = check_type_validity(timeout, [int, float], 30000)
self.wait = check_type_validity(wait, [int, float], 0) self.wait = check_type_validity(wait, [int, float], 0)
if page_action is not None: if page_action is not None:
@@ -337,6 +347,9 @@ class PlaywrightEngine:
browser = p.chromium.launch(**self.__launch_kwargs()) browser = p.chromium.launch(**self.__launch_kwargs())
context = browser.new_context(**self.__context_kwargs()) context = browser.new_context(**self.__context_kwargs())
if self.cookies:
context.add_cookies(self.cookies)
page = context.new_page() page = context.new_page()
page.set_default_navigation_timeout(self.timeout) page.set_default_navigation_timeout(self.timeout)
page.set_default_timeout(self.timeout) page.set_default_timeout(self.timeout)
@@ -449,6 +462,9 @@ class PlaywrightEngine:
browser = await p.chromium.launch(**self.__launch_kwargs()) browser = await p.chromium.launch(**self.__launch_kwargs())
context = await browser.new_context(**self.__context_kwargs()) context = await browser.new_context(**self.__context_kwargs())
if self.cookies:
await context.add_cookies(self.cookies)
page = await context.new_page() page = await context.new_page()
page.set_default_navigation_timeout(self.timeout) page.set_default_navigation_timeout(self.timeout)
page.set_default_timeout(self.timeout) page.set_default_timeout(self.timeout)
+13
View File
@@ -6,6 +6,7 @@ from scrapling.core._types import (
Optional, Optional,
SelectorWaitStates, SelectorWaitStates,
Union, Union,
Iterable,
) )
from scrapling.engines import ( from scrapling.engines import (
CamoufoxEngine, CamoufoxEngine,
@@ -484,6 +485,7 @@ class StealthyFetcher(BaseFetcher):
allow_webgl: bool = True, allow_webgl: bool = True,
network_idle: bool = False, network_idle: bool = False,
addons: Optional[List[str]] = None, addons: Optional[List[str]] = None,
cookies: Optional[Iterable[Dict]] = None,
wait: Optional[int] = 0, wait: Optional[int] = 0,
timeout: Optional[float] = 30000, timeout: Optional[float] = 30000,
page_action: Callable = None, page_action: Callable = None,
@@ -511,6 +513,7 @@ class StealthyFetcher(BaseFetcher):
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
:param block_webrtc: Blocks WebRTC entirely. :param block_webrtc: Blocks WebRTC entirely.
:param cookies: Set cookies for the next request.
:param addons: List of Firefox addons to use. Must be paths to extracted addons. :param addons: List of Firefox addons to use. Must be paths to extracted addons.
:param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window. :param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window.
:param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page before returning the response to you. :param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page before returning the response to you.
@@ -545,6 +548,7 @@ class StealthyFetcher(BaseFetcher):
geoip=geoip, geoip=geoip,
addons=addons, addons=addons,
timeout=timeout, timeout=timeout,
cookies=cookies,
headless=headless, headless=headless,
humanize=humanize, humanize=humanize,
disable_ads=disable_ads, disable_ads=disable_ads,
@@ -573,6 +577,7 @@ class StealthyFetcher(BaseFetcher):
block_images: bool = False, block_images: bool = False,
disable_resources: bool = False, disable_resources: bool = False,
block_webrtc: bool = False, block_webrtc: bool = False,
cookies: Optional[Iterable[Dict]] = None,
allow_webgl: bool = True, allow_webgl: bool = True,
network_idle: bool = False, network_idle: bool = False,
addons: Optional[List[str]] = None, addons: Optional[List[str]] = None,
@@ -603,6 +608,7 @@ class StealthyFetcher(BaseFetcher):
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
:param block_webrtc: Blocks WebRTC entirely. :param block_webrtc: Blocks WebRTC entirely.
:param cookies: Set cookies for the next request.
:param addons: List of Firefox addons to use. Must be paths to extracted addons. :param addons: List of Firefox addons to use. Must be paths to extracted addons.
:param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window. :param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window.
:param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page before returning the response to you. :param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page before returning the response to you.
@@ -637,6 +643,7 @@ class StealthyFetcher(BaseFetcher):
geoip=geoip, geoip=geoip,
addons=addons, addons=addons,
timeout=timeout, timeout=timeout,
cookies=cookies,
headless=headless, headless=headless,
humanize=humanize, humanize=humanize,
disable_ads=disable_ads, disable_ads=disable_ads,
@@ -685,6 +692,7 @@ class PlayWrightFetcher(BaseFetcher):
network_idle: bool = False, network_idle: bool = False,
timeout: Optional[float] = 30000, timeout: Optional[float] = 30000,
wait: Optional[int] = 0, wait: Optional[int] = 0,
cookies: Optional[Iterable[Dict]] = None,
page_action: Optional[Callable] = None, page_action: Optional[Callable] = None,
wait_selector: Optional[str] = None, wait_selector: Optional[str] = None,
wait_selector_state: SelectorWaitStates = "attached", wait_selector_state: SelectorWaitStates = "attached",
@@ -712,6 +720,7 @@ class PlayWrightFetcher(BaseFetcher):
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. :param network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 :param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. :param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
:param cookies: Set cookies for the next request.
:param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again. :param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again.
:param wait_selector: Wait for a specific CSS selector to be in a specific state. :param wait_selector: Wait for a specific CSS selector to be in a specific state.
:param locale: Set the locale for the browser if wanted. The default value is `en-US`. :param locale: Set the locale for the browser if wanted. The default value is `en-US`.
@@ -743,6 +752,7 @@ class PlayWrightFetcher(BaseFetcher):
timeout=timeout, timeout=timeout,
stealth=stealth, stealth=stealth,
cdp_url=cdp_url, cdp_url=cdp_url,
cookies=cookies,
headless=headless, headless=headless,
useragent=useragent, useragent=useragent,
real_chrome=real_chrome, real_chrome=real_chrome,
@@ -771,6 +781,7 @@ class PlayWrightFetcher(BaseFetcher):
network_idle: bool = False, network_idle: bool = False,
timeout: Optional[float] = 30000, timeout: Optional[float] = 30000,
wait: Optional[int] = 0, wait: Optional[int] = 0,
cookies: Optional[Iterable[Dict]] = None,
page_action: Optional[Callable] = None, page_action: Optional[Callable] = None,
wait_selector: Optional[str] = None, wait_selector: Optional[str] = None,
wait_selector_state: SelectorWaitStates = "attached", wait_selector_state: SelectorWaitStates = "attached",
@@ -796,6 +807,7 @@ class PlayWrightFetcher(BaseFetcher):
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
:param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it. :param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it.
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. :param network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param cookies: Set cookies for the next request.
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 :param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. :param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
:param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again. :param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again.
@@ -829,6 +841,7 @@ class PlayWrightFetcher(BaseFetcher):
timeout=timeout, timeout=timeout,
stealth=stealth, stealth=stealth,
cdp_url=cdp_url, cdp_url=cdp_url,
cookies=cookies,
headless=headless, headless=headless,
useragent=useragent, useragent=useragent,
real_chrome=real_chrome, real_chrome=real_chrome,