diff --git a/scrapling/engines/_browsers/_base.py b/scrapling/engines/_browsers/_base.py
index 9a6dfae..e01a820 100644
--- a/scrapling/engines/_browsers/_base.py
+++ b/scrapling/engines/_browsers/_base.py
@@ -53,7 +53,9 @@ class SyncSession:
for script in _compiled_stealth_scripts():
page.add_init_script(script=script)
- return self.page_pool.add_page(page)
+ page_info = self.page_pool.add_page(page)
+ page_info.mark_busy()
+ return page_info
def get_pool_stats(self) -> Dict[str, int]:
"""Get statistics about the current page pool"""
@@ -97,7 +99,6 @@ class AsyncSession:
f"No pages finished to clear place in the pool within the {self._max_wait_for_page}s timeout period"
)
- assert self.context is not None, "Browser context not initialized"
page = await self.context.new_page()
page.set_default_navigation_timeout(timeout)
page.set_default_timeout(timeout)
diff --git a/scrapling/engines/_browsers/_camoufox.py b/scrapling/engines/_browsers/_camoufox.py
index 94e8d37..e66a826 100644
--- a/scrapling/engines/_browsers/_camoufox.py
+++ b/scrapling/engines/_browsers/_camoufox.py
@@ -206,21 +206,6 @@ class StealthySession(StealthySessionMixin, SyncSession):
self._closed = True
- @staticmethod
- def _get_page_content(page: Page) -> str:
- """
- A workaround for the Playwright issue with `page.content()` on Windows. Ref.: https://github.com/microsoft/playwright/issues/16108
- :param page: The page to extract content from.
- :return:
- """
- while True:
- try:
- return page.content() or ""
- except PlaywrightError:
- page.wait_for_timeout(1000)
- continue
- return "" # pyright: ignore
-
def _solve_cloudflare(self, page: Page) -> None: # pragma: no cover
"""Solve the cloudflare challenge displayed on the playwright page passed
@@ -231,14 +216,14 @@ class StealthySession(StealthySessionMixin, SyncSession):
page.wait_for_load_state("networkidle", timeout=5000)
except PlaywrightError:
pass
- challenge_type = self._detect_cloudflare(self._get_page_content(page))
+ challenge_type = self._detect_cloudflare(ResponseFactory._get_page_content(page))
if not challenge_type:
log.error("No Cloudflare challenge found.")
return
else:
log.info(f'The turnstile version discovered is "{challenge_type}"')
if challenge_type == "non-interactive":
- while "
Just a moment..." in (self._get_page_content(page)):
+ while "Just a moment..." in (ResponseFactory._get_page_content(page)):
log.info("Waiting for Cloudflare wait page to disappear.")
page.wait_for_timeout(1000)
page.wait_for_load_state()
@@ -249,7 +234,7 @@ class StealthySession(StealthySessionMixin, SyncSession):
box_selector = "#cf_turnstile div, #cf-turnstile div, .turnstile>div>div"
if challenge_type != "embedded":
box_selector = ".main-content p+div>div>div"
- while "Verifying you are human." in self._get_page_content(page):
+ while "Verifying you are human." in ResponseFactory._get_page_content(page):
# Waiting for the verify spinner to disappear, checking every 1s if it disappeared
page.wait_for_timeout(500)
@@ -403,7 +388,7 @@ class StealthySession(StealthySessionMixin, SyncSession):
page_info.page.wait_for_timeout(params.wait)
response = ResponseFactory.from_playwright_response(
- page_info.page, first_response, final_response, params.selector_config
+ page_info.page, first_response, final_response, params.selector_config, bool(params.page_action)
)
# Close the page to free up resources
@@ -550,21 +535,6 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
self._closed = True
- @staticmethod
- async def _get_page_content(page: async_Page) -> str:
- """
- A workaround for the Playwright issue with `page.content()` on Windows. Ref.: https://github.com/microsoft/playwright/issues/16108
- :param page: The page to extract content from.
- :return:
- """
- while True:
- try:
- return (await page.content()) or ""
- except PlaywrightError:
- await page.wait_for_timeout(1000)
- continue
- return "" # pyright: ignore
-
async def _solve_cloudflare(self, page: async_Page):
"""Solve the cloudflare challenge displayed on the playwright page passed. The async version
@@ -575,14 +545,14 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
await page.wait_for_load_state("networkidle", timeout=5000)
except PlaywrightError:
pass
- challenge_type = self._detect_cloudflare(await self._get_page_content(page))
+ challenge_type = self._detect_cloudflare(await ResponseFactory._get_async_page_content(page))
if not challenge_type:
log.error("No Cloudflare challenge found.")
return
else:
log.info(f'The turnstile version discovered is "{challenge_type}"')
if challenge_type == "non-interactive": # pragma: no cover
- while "Just a moment..." in (await self._get_page_content(page)):
+ while "Just a moment..." in (await ResponseFactory._get_async_page_content(page)):
log.info("Waiting for Cloudflare wait page to disappear.")
await page.wait_for_timeout(1000)
await page.wait_for_load_state()
@@ -593,7 +563,7 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
box_selector = "#cf_turnstile div, #cf-turnstile div, .turnstile>div>div"
if challenge_type != "embedded":
box_selector = ".main-content p+div>div>div"
- while "Verifying you are human." in (await self._get_page_content(page)):
+ while "Verifying you are human." in (await ResponseFactory._get_async_page_content(page)):
# Waiting for the verify spinner to disappear, checking every 1s if it disappeared
await page.wait_for_timeout(500)
@@ -753,7 +723,7 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
# Create response object
response = await ResponseFactory.from_async_playwright_response(
- page_info.page, first_response, final_response, params.selector_config
+ page_info.page, first_response, final_response, params.selector_config, bool(params.page_action)
)
# Close the page to free up resources
diff --git a/scrapling/engines/_browsers/_controllers.py b/scrapling/engines/_browsers/_controllers.py
index ca8b45e..45a6938 100644
--- a/scrapling/engines/_browsers/_controllers.py
+++ b/scrapling/engines/_browsers/_controllers.py
@@ -306,7 +306,7 @@ class DynamicSession(DynamicSessionMixin, SyncSession):
# Create response object
response = ResponseFactory.from_playwright_response(
- page_info.page, first_response, final_response, params.selector_config
+ page_info.page, first_response, final_response, params.selector_config, bool(params.page_action)
)
# Close the page to free up resources
@@ -563,7 +563,7 @@ class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
# Create response object
response = await ResponseFactory.from_async_playwright_response(
- page_info.page, first_response, final_response, params.selector_config
+ page_info.page, first_response, final_response, params.selector_config, bool(params.page_action)
)
# Close the page to free up resources
diff --git a/scrapling/engines/toolbelt/convertor.py b/scrapling/engines/toolbelt/convertor.py
index b66b518..5bb4a49 100644
--- a/scrapling/engines/toolbelt/convertor.py
+++ b/scrapling/engines/toolbelt/convertor.py
@@ -2,6 +2,7 @@ from functools import lru_cache
from re import compile as re_compile
from curl_cffi.requests import Response as CurlResponse
+from playwright._impl._errors import Error as PlaywrightError
from playwright.sync_api import Page as SyncPage, Response as SyncResponse
from playwright.async_api import Page as AsyncPage, Response as AsyncResponse
@@ -84,6 +85,7 @@ class ResponseFactory:
first_response: SyncResponse,
final_response: Optional[SyncResponse],
parser_arguments: Dict,
+ automated_page: bool = False,
) -> Response:
"""
Transforms a Playwright response into an internal `Response` object, encapsulating
@@ -99,6 +101,7 @@ class ResponseFactory:
:param first_response: An earlier or initial Playwright `Response` object that may serve as a fallback response in the absence of the final one.
:param parser_arguments: A dictionary containing additional arguments needed for parsing or further customization of the returned `Response`. These arguments are dynamically unpacked into
the `Response` object.
+ :param automated_page: If True, it means the `page_action` argument was being used, so the response retrieving method changes to use Playwright's page instead of the final response.
:return: A fully populated `Response` object containing the page's URL, content, status, headers, cookies, and other derived metadata.
:rtype: Response
@@ -114,7 +117,7 @@ class ResponseFactory:
history = cls._process_response_history(first_response, parser_arguments)
try:
- page_content = final_response.text()
+ page_content = final_response.text() if not automated_page else cls._get_page_content(page)
except Exception as e: # pragma: no cover
log.error(f"Error getting page content: {e}")
page_content = ""
@@ -179,6 +182,36 @@ class ResponseFactory:
return history
+ @classmethod
+ def _get_page_content(cls, page: SyncPage) -> str:
+ """
+ A workaround for the Playwright issue with `page.content()` on Windows. Ref.: https://github.com/microsoft/playwright/issues/16108
+ :param page: The page to extract content from.
+ :return:
+ """
+ while True:
+ try:
+ return page.content() or ""
+ except PlaywrightError:
+ page.wait_for_timeout(500)
+ continue
+ return "" # pyright: ignore
+
+ @classmethod
+ async def _get_async_page_content(cls, page: AsyncPage) -> str:
+ """
+ A workaround for the Playwright issue with `page.content()` on Windows. Ref.: https://github.com/microsoft/playwright/issues/16108
+ :param page: The page to extract content from.
+ :return:
+ """
+ while True:
+ try:
+ return (await page.content()) or ""
+ except PlaywrightError:
+ await page.wait_for_timeout(500)
+ continue
+ return "" # pyright: ignore
+
@classmethod
async def from_async_playwright_response(
cls,
@@ -186,6 +219,7 @@ class ResponseFactory:
first_response: AsyncResponse,
final_response: Optional[AsyncResponse],
parser_arguments: Dict,
+ automated_page: bool = False,
) -> Response:
"""
Transforms a Playwright response into an internal `Response` object, encapsulating
@@ -201,6 +235,7 @@ class ResponseFactory:
:param first_response: An earlier or initial Playwright `Response` object that may serve as a fallback response in the absence of the final one.
:param parser_arguments: A dictionary containing additional arguments needed for parsing or further customization of the returned `Response`. These arguments are dynamically unpacked into
the `Response` object.
+ :param automated_page: If True, it means the `page_action` argument was being used, so the response retrieving method changes to use Playwright's page instead of the final response.
:return: A fully populated `Response` object containing the page's URL, content, status, headers, cookies, and other derived metadata.
:rtype: Response
@@ -216,7 +251,7 @@ class ResponseFactory:
history = await cls._async_process_response_history(first_response, parser_arguments)
try:
- page_content = await final_response.text()
+ page_content = await (final_response.text() if not automated_page else cls._get_async_page_content(page))
except Exception as e: # pragma: no cover
log.error(f"Error getting page content in async: {e}")
page_content = ""