feat(StealthyFetcher): adding the option to solve Cloudflare Turnstile
This commit is contained in:
+159
-12
@@ -1,6 +1,10 @@
|
|||||||
|
import re
|
||||||
|
|
||||||
from camoufox import DefaultAddons
|
from camoufox import DefaultAddons
|
||||||
from camoufox.async_api import AsyncCamoufox
|
from playwright.sync_api import Page
|
||||||
from camoufox.sync_api import Camoufox
|
from camoufox.sync_api import Camoufox
|
||||||
|
from camoufox.async_api import AsyncCamoufox
|
||||||
|
from playwright.async_api import Page as async_Page
|
||||||
|
|
||||||
from scrapling.core._types import (
|
from scrapling.core._types import (
|
||||||
Callable,
|
Callable,
|
||||||
@@ -34,6 +38,7 @@ class CamoufoxEngine:
|
|||||||
allow_webgl: bool = True,
|
allow_webgl: bool = True,
|
||||||
network_idle: bool = False,
|
network_idle: bool = False,
|
||||||
humanize: Union[bool, float] = True,
|
humanize: Union[bool, float] = True,
|
||||||
|
solve_cloudflare: Optional[bool] = False,
|
||||||
wait: Optional[int] = 0,
|
wait: Optional[int] = 0,
|
||||||
timeout: Optional[float] = 30000,
|
timeout: Optional[float] = 30000,
|
||||||
page_action: Callable = None,
|
page_action: Callable = None,
|
||||||
@@ -49,7 +54,7 @@ class CamoufoxEngine:
|
|||||||
adaptor_arguments: Dict = None,
|
adaptor_arguments: Dict = None,
|
||||||
additional_arguments: Dict = None,
|
additional_arguments: Dict = None,
|
||||||
):
|
):
|
||||||
"""An engine that utilizes Camoufox library, check the `StealthyFetcher` class for more documentation.
|
"""An engine that uses the Camoufox library; Check the `StealthyFetcher` class for more documentation.
|
||||||
|
|
||||||
:param headless: Run the browser in headless/hidden (default), virtual screen mode, or headful/visible mode.
|
:param headless: Run the browser in headless/hidden (default), virtual screen mode, or headful/visible mode.
|
||||||
:param block_images: Prevent the loading of images through Firefox preferences.
|
:param block_images: Prevent the loading of images through Firefox preferences.
|
||||||
@@ -60,22 +65,23 @@ class CamoufoxEngine:
|
|||||||
:param block_webrtc: Blocks WebRTC entirely.
|
:param block_webrtc: Blocks WebRTC entirely.
|
||||||
:param addons: List of Firefox addons to use. Must be paths to extracted addons.
|
:param addons: List of Firefox addons to use. Must be paths to extracted addons.
|
||||||
:param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window.
|
:param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window.
|
||||||
:param allow_webgl: Enabled by default. Disabling it WebGL not recommended as many WAFs now checks if WebGL is enabled.
|
:param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page before returning the response to you.
|
||||||
|
:param allow_webgl: Enabled by default. Disabling WebGL is not recommended as many WAFs now check if WebGL is enabled.
|
||||||
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
||||||
:param disable_ads: Disabled by default, this installs `uBlock Origin` addon on the browser if enabled.
|
:param disable_ads: Disabled by default, this installs the `uBlock Origin` addon on the browser if enabled.
|
||||||
:param os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. The default is Scrapling matching the fingerprints with the current OS.
|
:param os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. The default is Scrapling matching the fingerprints with the current OS.
|
||||||
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning `Response` object.
|
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
|
||||||
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30000
|
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
|
||||||
:param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again.
|
:param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again.
|
||||||
:param wait_selector: Wait for a specific css selector to be in a specific state.
|
:param wait_selector: Wait for a specific css selector to be in a specific state.
|
||||||
:param geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, & spoof the WebRTC IP address.
|
:param geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, and spoof the WebRTC IP address.
|
||||||
It will also calculate and spoof the browser's language based on the distribution of language speakers in the target region.
|
It will also calculate and spoof the browser's language based on the distribution of language speakers in the target region.
|
||||||
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. Default state is `attached`.
|
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
|
||||||
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search for this website's domain name.
|
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search for this website's domain name.
|
||||||
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
|
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
|
||||||
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
|
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
|
||||||
:param adaptor_arguments: The arguments that will be passed in the end while creating the final Adaptor's class.
|
:param adaptor_arguments: The arguments that will be passed in the end while creating the final Adaptor's class.
|
||||||
:param additional_arguments: Additional arguments to be passed to Camoufox as additional settings and it takes higher priority than Scrapling's settings.
|
:param additional_arguments: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings.
|
||||||
"""
|
"""
|
||||||
self.headless = headless
|
self.headless = headless
|
||||||
self.block_images = bool(block_images)
|
self.block_images = bool(block_images)
|
||||||
@@ -92,9 +98,13 @@ class CamoufoxEngine:
|
|||||||
self.proxy = construct_proxy_dict(proxy)
|
self.proxy = construct_proxy_dict(proxy)
|
||||||
self.addons = addons or []
|
self.addons = addons or []
|
||||||
self.humanize = humanize
|
self.humanize = humanize
|
||||||
self.timeout = check_type_validity(timeout, [int, float], 30000)
|
self.solve_cloudflare = solve_cloudflare
|
||||||
|
self.timeout = check_type_validity(timeout, [int, float], 30_000)
|
||||||
self.wait = check_type_validity(wait, [int, float], 0)
|
self.wait = check_type_validity(wait, [int, float], 0)
|
||||||
|
|
||||||
|
if self.solve_cloudflare and self.timeout < 60_000:
|
||||||
|
self.timeout = 60_000
|
||||||
|
|
||||||
# Page action callable validation
|
# Page action callable validation
|
||||||
self.page_action = None
|
self.page_action = None
|
||||||
if page_action is not None:
|
if page_action is not None:
|
||||||
@@ -109,6 +119,10 @@ class CamoufoxEngine:
|
|||||||
|
|
||||||
def _get_camoufox_options(self):
|
def _get_camoufox_options(self):
|
||||||
"""Return consistent browser options dictionary for both sync and async methods"""
|
"""Return consistent browser options dictionary for both sync and async methods"""
|
||||||
|
humanize = self.humanize
|
||||||
|
if self.solve_cloudflare:
|
||||||
|
humanize = True
|
||||||
|
|
||||||
return {
|
return {
|
||||||
"geoip": self.geoip,
|
"geoip": self.geoip,
|
||||||
"proxy": self.proxy,
|
"proxy": self.proxy,
|
||||||
@@ -116,11 +130,11 @@ class CamoufoxEngine:
|
|||||||
"addons": self.addons,
|
"addons": self.addons,
|
||||||
"exclude_addons": [] if self.disable_ads else [DefaultAddons.UBO],
|
"exclude_addons": [] if self.disable_ads else [DefaultAddons.UBO],
|
||||||
"headless": self.headless,
|
"headless": self.headless,
|
||||||
"humanize": self.humanize,
|
"humanize": humanize,
|
||||||
"i_know_what_im_doing": True, # To turn warnings off with the user configurations
|
"i_know_what_im_doing": True, # To turn warnings off with the user configurations
|
||||||
"allow_webgl": self.allow_webgl,
|
"allow_webgl": self.allow_webgl,
|
||||||
"block_webrtc": self.block_webrtc,
|
"block_webrtc": self.block_webrtc,
|
||||||
"block_images": self.block_images, # Careful! it makes some websites doesn't finish loading at all like stackoverflow even in headful
|
"block_images": self.block_images, # Careful! it makes some websites don't finish loading at all like stackoverflow even in headful mode.
|
||||||
"os": None if self.os_randomize else get_os_name(),
|
"os": None if self.os_randomize else get_os_name(),
|
||||||
**self.additional_arguments,
|
**self.additional_arguments,
|
||||||
}
|
}
|
||||||
@@ -211,6 +225,123 @@ class CamoufoxEngine:
|
|||||||
|
|
||||||
return history
|
return history
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __detect_cloudflare(page_content):
|
||||||
|
challenge_types = (
|
||||||
|
"non-interactive",
|
||||||
|
"managed",
|
||||||
|
"interactive",
|
||||||
|
)
|
||||||
|
for ctype in challenge_types:
|
||||||
|
if f"cType: '{ctype}'" in page_content:
|
||||||
|
return ctype
|
||||||
|
|
||||||
|
return None
|
||||||
|
|
||||||
|
def _solve_cloudflare(self, page: Page) -> None:
|
||||||
|
"""Solve the cloudflare challenge displayed on the playwright page passed
|
||||||
|
|
||||||
|
:param page: The targeted page
|
||||||
|
:return:
|
||||||
|
"""
|
||||||
|
page_content = page.content()
|
||||||
|
challenge_type = self.__detect_cloudflare(page_content)
|
||||||
|
if not challenge_type:
|
||||||
|
log.error("No Cloudflare challenge found.")
|
||||||
|
return
|
||||||
|
else:
|
||||||
|
log.info(f'The turnstile version discovered is "{challenge_type}"')
|
||||||
|
if challenge_type == "non-interactive":
|
||||||
|
while "<title>Just a moment...</title>" in (page.content()):
|
||||||
|
log.info("Waiting for Cloudflare wait page to disappear.")
|
||||||
|
page.wait_for_timeout(1000)
|
||||||
|
page.wait_for_load_state()
|
||||||
|
log.info("Cloudflare captcha is solved")
|
||||||
|
return
|
||||||
|
|
||||||
|
else:
|
||||||
|
while "Verifying you are human." in page.content():
|
||||||
|
# Waiting for the verify spinner to disappear, checking every 1s if it disappeared
|
||||||
|
page.wait_for_timeout(1000)
|
||||||
|
|
||||||
|
iframe = page.frame(
|
||||||
|
url=re.compile(
|
||||||
|
"challenges.cloudflare.com/cdn-cgi/challenge-platform/.*"
|
||||||
|
)
|
||||||
|
)
|
||||||
|
if iframe is None:
|
||||||
|
print("No iframe bro")
|
||||||
|
return
|
||||||
|
|
||||||
|
while not iframe.frame_element().is_visible():
|
||||||
|
# Double-checking that the iframe is loaded
|
||||||
|
page.wait_for_timeout(1000)
|
||||||
|
|
||||||
|
# Calculate the Captcha coordinates for any viewport
|
||||||
|
outer_box = page.locator(".main-content p+div>div>div").bounding_box()
|
||||||
|
captcha_x, captcha_y = outer_box["x"] + 26, outer_box["y"] + 25
|
||||||
|
|
||||||
|
# Move the mouse to the center of the window, then press and hold the left mouse button
|
||||||
|
page.mouse.click(captcha_x, captcha_y, delay=60, button="left")
|
||||||
|
page.locator(".zone-name-title").wait_for(state="hidden")
|
||||||
|
page.wait_for_load_state(state="domcontentloaded")
|
||||||
|
|
||||||
|
log.info("Cloudflare captcha is solved")
|
||||||
|
return
|
||||||
|
|
||||||
|
async def _async_solve_cloudflare(self, page: async_Page):
|
||||||
|
"""Solve the cloudflare challenge displayed on the playwright page passed. The async version
|
||||||
|
|
||||||
|
:param page: The async targeted page
|
||||||
|
:return:
|
||||||
|
"""
|
||||||
|
page_content = await page.content()
|
||||||
|
challenge_type = self.__detect_cloudflare(page_content)
|
||||||
|
if not challenge_type:
|
||||||
|
log.error("No Cloudflare challenge found.")
|
||||||
|
return
|
||||||
|
else:
|
||||||
|
log.info(f'The turnstile version discovered is "{challenge_type}"')
|
||||||
|
if challenge_type == "non-interactive":
|
||||||
|
while "<title>Just a moment...</title>" in (await page.content()):
|
||||||
|
log.info("Waiting for Cloudflare wait page to disappear.")
|
||||||
|
await page.wait_for_timeout(1000)
|
||||||
|
await page.wait_for_load_state()
|
||||||
|
log.info("Cloudflare captcha is solved")
|
||||||
|
return
|
||||||
|
|
||||||
|
else:
|
||||||
|
while "Verifying you are human." in (await page.content()):
|
||||||
|
# Waiting for the verify spinner to disappear, checking every 1s if it disappeared
|
||||||
|
await page.wait_for_timeout(1000)
|
||||||
|
|
||||||
|
iframe = page.frame(
|
||||||
|
url=re.compile(
|
||||||
|
"challenges.cloudflare.com/cdn-cgi/challenge-platform/.*"
|
||||||
|
)
|
||||||
|
)
|
||||||
|
if iframe is None:
|
||||||
|
print("No iframe bro")
|
||||||
|
return
|
||||||
|
|
||||||
|
while not await (await iframe.frame_element()).is_visible():
|
||||||
|
# Double-checking that the iframe is loaded
|
||||||
|
await page.wait_for_timeout(1000)
|
||||||
|
|
||||||
|
# Calculate the Captcha coordinates for any viewport
|
||||||
|
outer_box = await page.locator(
|
||||||
|
".main-content p+div>div>div"
|
||||||
|
).bounding_box()
|
||||||
|
captcha_x, captcha_y = outer_box["x"] + 26, outer_box["y"] + 25
|
||||||
|
|
||||||
|
# Move the mouse to the center of the window, then press and hold the left mouse button
|
||||||
|
await page.mouse.click(captcha_x, captcha_y, delay=60, button="left")
|
||||||
|
await page.locator(".zone-name-title").wait_for(state="hidden")
|
||||||
|
await page.wait_for_load_state(state="domcontentloaded")
|
||||||
|
|
||||||
|
log.info("Cloudflare captcha is solved")
|
||||||
|
return
|
||||||
|
|
||||||
def fetch(self, url: str) -> Response:
|
def fetch(self, url: str) -> Response:
|
||||||
"""Opens up the browser and do your request based on your chosen options.
|
"""Opens up the browser and do your request based on your chosen options.
|
||||||
|
|
||||||
@@ -247,6 +378,14 @@ class CamoufoxEngine:
|
|||||||
if self.network_idle:
|
if self.network_idle:
|
||||||
page.wait_for_load_state("networkidle")
|
page.wait_for_load_state("networkidle")
|
||||||
|
|
||||||
|
if self.solve_cloudflare:
|
||||||
|
self._solve_cloudflare(page)
|
||||||
|
# Make sure the page is fully loaded after the captcha
|
||||||
|
page.wait_for_load_state(state="load")
|
||||||
|
page.wait_for_load_state(state="domcontentloaded")
|
||||||
|
if self.network_idle:
|
||||||
|
page.wait_for_load_state("networkidle")
|
||||||
|
|
||||||
if self.page_action is not None:
|
if self.page_action is not None:
|
||||||
try:
|
try:
|
||||||
page = self.page_action(page)
|
page = self.page_action(page)
|
||||||
@@ -343,6 +482,14 @@ class CamoufoxEngine:
|
|||||||
if self.network_idle:
|
if self.network_idle:
|
||||||
await page.wait_for_load_state("networkidle")
|
await page.wait_for_load_state("networkidle")
|
||||||
|
|
||||||
|
if self.solve_cloudflare:
|
||||||
|
await self._async_solve_cloudflare(page)
|
||||||
|
# Make sure the page is fully loaded after the captcha
|
||||||
|
await page.wait_for_load_state(state="load")
|
||||||
|
await page.wait_for_load_state(state="domcontentloaded")
|
||||||
|
if self.network_idle:
|
||||||
|
await page.wait_for_load_state("networkidle")
|
||||||
|
|
||||||
if self.page_action is not None:
|
if self.page_action is not None:
|
||||||
try:
|
try:
|
||||||
page = await self.page_action(page)
|
page = await self.page_action(page)
|
||||||
|
|||||||
+24
-18
@@ -489,6 +489,7 @@ class StealthyFetcher(BaseFetcher):
|
|||||||
page_action: Callable = None,
|
page_action: Callable = None,
|
||||||
wait_selector: Optional[str] = None,
|
wait_selector: Optional[str] = None,
|
||||||
humanize: Optional[Union[bool, float]] = True,
|
humanize: Optional[Union[bool, float]] = True,
|
||||||
|
solve_cloudflare: Optional[bool] = False,
|
||||||
wait_selector_state: SelectorWaitStates = "attached",
|
wait_selector_state: SelectorWaitStates = "attached",
|
||||||
google_search: bool = True,
|
google_search: bool = True,
|
||||||
extra_headers: Optional[Dict[str, str]] = None,
|
extra_headers: Optional[Dict[str, str]] = None,
|
||||||
@@ -503,7 +504,7 @@ class StealthyFetcher(BaseFetcher):
|
|||||||
Opens up a browser and do your request based on your chosen options below.
|
Opens up a browser and do your request based on your chosen options below.
|
||||||
|
|
||||||
:param url: Target url.
|
:param url: Target url.
|
||||||
:param headless: Run the browser in headless/hidden (default), 'virtual' screen mode, or headful/visible mode.
|
:param headless: Run the browser in headless/hidden (default), virtual screen mode, or headful/visible mode.
|
||||||
:param block_images: Prevent the loading of images through Firefox preferences.
|
:param block_images: Prevent the loading of images through Firefox preferences.
|
||||||
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
|
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
|
||||||
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends but it made requests ~25% faster in my tests for some websites.
|
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends but it made requests ~25% faster in my tests for some websites.
|
||||||
@@ -511,23 +512,24 @@ class StealthyFetcher(BaseFetcher):
|
|||||||
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
|
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
|
||||||
:param block_webrtc: Blocks WebRTC entirely.
|
:param block_webrtc: Blocks WebRTC entirely.
|
||||||
:param addons: List of Firefox addons to use. Must be paths to extracted addons.
|
:param addons: List of Firefox addons to use. Must be paths to extracted addons.
|
||||||
:param disable_ads: Disabled by default, this installs `uBlock Origin` addon on the browser if enabled.
|
|
||||||
:param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window.
|
:param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window.
|
||||||
:param allow_webgl: Enabled by default. Disabling it WebGL not recommended as many WAFs now checks if WebGL is enabled.
|
:param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page before returning the response to you.
|
||||||
:param geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, & spoof the WebRTC IP address.
|
:param allow_webgl: Enabled by default. Disabling WebGL is not recommended as many WAFs now check if WebGL is enabled.
|
||||||
It will also calculate and spoof the browser's language based on the distribution of language speakers in the target region.
|
|
||||||
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
||||||
|
:param disable_ads: Disabled by default, this installs the `uBlock Origin` addon on the browser if enabled.
|
||||||
:param os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. The default is Scrapling matching the fingerprints with the current OS.
|
:param os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. The default is Scrapling matching the fingerprints with the current OS.
|
||||||
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30000.
|
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
|
||||||
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning `Response` object.
|
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
|
||||||
:param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again.
|
:param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again.
|
||||||
:param wait_selector: Wait for a specific css selector to be in a specific state.
|
:param wait_selector: Wait for a specific css selector to be in a specific state.
|
||||||
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. Default state is `attached`.
|
:param geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, and spoof the WebRTC IP address.
|
||||||
|
It will also calculate and spoof the browser's language based on the distribution of language speakers in the target region.
|
||||||
|
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
|
||||||
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search for this website's domain name.
|
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search for this website's domain name.
|
||||||
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
|
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
|
||||||
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
|
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
|
||||||
:param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values.
|
:param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values.
|
||||||
:param additional_arguments: Additional arguments to be passed to Camoufox as additional settings and it takes higher priority than Scrapling's settings.
|
:param additional_arguments: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings.
|
||||||
:return: A `Response` object that is the same as `Adaptor` object except it has these added attributes: `status`, `reason`, `cookies`, `headers`, and `request_headers`
|
:return: A `Response` object that is the same as `Adaptor` object except it has these added attributes: `status`, `reason`, `cookies`, `headers`, and `request_headers`
|
||||||
"""
|
"""
|
||||||
if not custom_config:
|
if not custom_config:
|
||||||
@@ -555,6 +557,7 @@ class StealthyFetcher(BaseFetcher):
|
|||||||
wait_selector=wait_selector,
|
wait_selector=wait_selector,
|
||||||
google_search=google_search,
|
google_search=google_search,
|
||||||
extra_headers=extra_headers,
|
extra_headers=extra_headers,
|
||||||
|
solve_cloudflare=solve_cloudflare,
|
||||||
disable_resources=disable_resources,
|
disable_resources=disable_resources,
|
||||||
wait_selector_state=wait_selector_state,
|
wait_selector_state=wait_selector_state,
|
||||||
adaptor_arguments={**cls._generate_parser_arguments(), **custom_config},
|
adaptor_arguments={**cls._generate_parser_arguments(), **custom_config},
|
||||||
@@ -578,6 +581,7 @@ class StealthyFetcher(BaseFetcher):
|
|||||||
page_action: Callable = None,
|
page_action: Callable = None,
|
||||||
wait_selector: Optional[str] = None,
|
wait_selector: Optional[str] = None,
|
||||||
humanize: Optional[Union[bool, float]] = True,
|
humanize: Optional[Union[bool, float]] = True,
|
||||||
|
solve_cloudflare: Optional[bool] = False,
|
||||||
wait_selector_state: SelectorWaitStates = "attached",
|
wait_selector_state: SelectorWaitStates = "attached",
|
||||||
google_search: bool = True,
|
google_search: bool = True,
|
||||||
extra_headers: Optional[Dict[str, str]] = None,
|
extra_headers: Optional[Dict[str, str]] = None,
|
||||||
@@ -592,7 +596,7 @@ class StealthyFetcher(BaseFetcher):
|
|||||||
Opens up a browser and do your request based on your chosen options below.
|
Opens up a browser and do your request based on your chosen options below.
|
||||||
|
|
||||||
:param url: Target url.
|
:param url: Target url.
|
||||||
:param headless: Run the browser in headless/hidden (default), 'virtual' screen mode, or headful/visible mode.
|
:param headless: Run the browser in headless/hidden (default), virtual screen mode, or headful/visible mode.
|
||||||
:param block_images: Prevent the loading of images through Firefox preferences.
|
:param block_images: Prevent the loading of images through Firefox preferences.
|
||||||
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
|
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
|
||||||
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends but it made requests ~25% faster in my tests for some websites.
|
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends but it made requests ~25% faster in my tests for some websites.
|
||||||
@@ -600,23 +604,24 @@ class StealthyFetcher(BaseFetcher):
|
|||||||
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
|
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
|
||||||
:param block_webrtc: Blocks WebRTC entirely.
|
:param block_webrtc: Blocks WebRTC entirely.
|
||||||
:param addons: List of Firefox addons to use. Must be paths to extracted addons.
|
:param addons: List of Firefox addons to use. Must be paths to extracted addons.
|
||||||
:param disable_ads: Disabled by default, this installs `uBlock Origin` addon on the browser if enabled.
|
|
||||||
:param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window.
|
:param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window.
|
||||||
:param allow_webgl: Enabled by default. Disabling it WebGL not recommended as many WAFs now checks if WebGL is enabled.
|
:param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page before returning the response to you.
|
||||||
:param geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, & spoof the WebRTC IP address.
|
:param allow_webgl: Enabled by default. Disabling WebGL is not recommended as many WAFs now check if WebGL is enabled.
|
||||||
It will also calculate and spoof the browser's language based on the distribution of language speakers in the target region.
|
|
||||||
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
||||||
|
:param disable_ads: Disabled by default, this installs the `uBlock Origin` addon on the browser if enabled.
|
||||||
:param os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. The default is Scrapling matching the fingerprints with the current OS.
|
:param os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. The default is Scrapling matching the fingerprints with the current OS.
|
||||||
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30000
|
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
|
||||||
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning `Response` object.
|
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
|
||||||
:param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again.
|
:param page_action: Added for automation. A function that takes the `page` object, does the automation you need, then returns `page` again.
|
||||||
:param wait_selector: Wait for a specific css selector to be in a specific state.
|
:param wait_selector: Wait for a specific css selector to be in a specific state.
|
||||||
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. Default state is `attached`.
|
:param geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, and spoof the WebRTC IP address.
|
||||||
|
It will also calculate and spoof the browser's language based on the distribution of language speakers in the target region.
|
||||||
|
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
|
||||||
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search for this website's domain name.
|
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search for this website's domain name.
|
||||||
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
|
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
|
||||||
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
|
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
|
||||||
:param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values.
|
:param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values.
|
||||||
:param additional_arguments: Additional arguments to be passed to Camoufox as additional settings and it takes higher priority than Scrapling's settings.
|
:param additional_arguments: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings.
|
||||||
:return: A `Response` object that is the same as `Adaptor` object except it has these added attributes: `status`, `reason`, `cookies`, `headers`, and `request_headers`
|
:return: A `Response` object that is the same as `Adaptor` object except it has these added attributes: `status`, `reason`, `cookies`, `headers`, and `request_headers`
|
||||||
"""
|
"""
|
||||||
if not custom_config:
|
if not custom_config:
|
||||||
@@ -644,6 +649,7 @@ class StealthyFetcher(BaseFetcher):
|
|||||||
wait_selector=wait_selector,
|
wait_selector=wait_selector,
|
||||||
google_search=google_search,
|
google_search=google_search,
|
||||||
extra_headers=extra_headers,
|
extra_headers=extra_headers,
|
||||||
|
solve_cloudflare=solve_cloudflare,
|
||||||
disable_resources=disable_resources,
|
disable_resources=disable_resources,
|
||||||
wait_selector_state=wait_selector_state,
|
wait_selector_state=wait_selector_state,
|
||||||
adaptor_arguments={**cls._generate_parser_arguments(), **custom_config},
|
adaptor_arguments={**cls._generate_parser_arguments(), **custom_config},
|
||||||
|
|||||||
Reference in New Issue
Block a user