diff --git a/scrapling/engines/_browsers/_base.py b/scrapling/engines/_browsers/_base.py index 9a6dfae..e01a820 100644 --- a/scrapling/engines/_browsers/_base.py +++ b/scrapling/engines/_browsers/_base.py @@ -53,7 +53,9 @@ class SyncSession: for script in _compiled_stealth_scripts(): page.add_init_script(script=script) - return self.page_pool.add_page(page) + page_info = self.page_pool.add_page(page) + page_info.mark_busy() + return page_info def get_pool_stats(self) -> Dict[str, int]: """Get statistics about the current page pool""" @@ -97,7 +99,6 @@ class AsyncSession: f"No pages finished to clear place in the pool within the {self._max_wait_for_page}s timeout period" ) - assert self.context is not None, "Browser context not initialized" page = await self.context.new_page() page.set_default_navigation_timeout(timeout) page.set_default_timeout(timeout) diff --git a/scrapling/engines/_browsers/_camoufox.py b/scrapling/engines/_browsers/_camoufox.py index 94e8d37..e66a826 100644 --- a/scrapling/engines/_browsers/_camoufox.py +++ b/scrapling/engines/_browsers/_camoufox.py @@ -206,21 +206,6 @@ class StealthySession(StealthySessionMixin, SyncSession): self._closed = True - @staticmethod - def _get_page_content(page: Page) -> str: - """ - A workaround for the Playwright issue with `page.content()` on Windows. Ref.: https://github.com/microsoft/playwright/issues/16108 - :param page: The page to extract content from. - :return: - """ - while True: - try: - return page.content() or "" - except PlaywrightError: - page.wait_for_timeout(1000) - continue - return "" # pyright: ignore - def _solve_cloudflare(self, page: Page) -> None: # pragma: no cover """Solve the cloudflare challenge displayed on the playwright page passed @@ -231,14 +216,14 @@ class StealthySession(StealthySessionMixin, SyncSession): page.wait_for_load_state("networkidle", timeout=5000) except PlaywrightError: pass - challenge_type = self._detect_cloudflare(self._get_page_content(page)) + challenge_type = self._detect_cloudflare(ResponseFactory._get_page_content(page)) if not challenge_type: log.error("No Cloudflare challenge found.") return else: log.info(f'The turnstile version discovered is "{challenge_type}"') if challenge_type == "non-interactive": - while "Just a moment..." in (self._get_page_content(page)): + while "Just a moment..." in (ResponseFactory._get_page_content(page)): log.info("Waiting for Cloudflare wait page to disappear.") page.wait_for_timeout(1000) page.wait_for_load_state() @@ -249,7 +234,7 @@ class StealthySession(StealthySessionMixin, SyncSession): box_selector = "#cf_turnstile div, #cf-turnstile div, .turnstile>div>div" if challenge_type != "embedded": box_selector = ".main-content p+div>div>div" - while "Verifying you are human." in self._get_page_content(page): + while "Verifying you are human." in ResponseFactory._get_page_content(page): # Waiting for the verify spinner to disappear, checking every 1s if it disappeared page.wait_for_timeout(500) @@ -403,7 +388,7 @@ class StealthySession(StealthySessionMixin, SyncSession): page_info.page.wait_for_timeout(params.wait) response = ResponseFactory.from_playwright_response( - page_info.page, first_response, final_response, params.selector_config + page_info.page, first_response, final_response, params.selector_config, bool(params.page_action) ) # Close the page to free up resources @@ -550,21 +535,6 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession): self._closed = True - @staticmethod - async def _get_page_content(page: async_Page) -> str: - """ - A workaround for the Playwright issue with `page.content()` on Windows. Ref.: https://github.com/microsoft/playwright/issues/16108 - :param page: The page to extract content from. - :return: - """ - while True: - try: - return (await page.content()) or "" - except PlaywrightError: - await page.wait_for_timeout(1000) - continue - return "" # pyright: ignore - async def _solve_cloudflare(self, page: async_Page): """Solve the cloudflare challenge displayed on the playwright page passed. The async version @@ -575,14 +545,14 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession): await page.wait_for_load_state("networkidle", timeout=5000) except PlaywrightError: pass - challenge_type = self._detect_cloudflare(await self._get_page_content(page)) + challenge_type = self._detect_cloudflare(await ResponseFactory._get_async_page_content(page)) if not challenge_type: log.error("No Cloudflare challenge found.") return else: log.info(f'The turnstile version discovered is "{challenge_type}"') if challenge_type == "non-interactive": # pragma: no cover - while "Just a moment..." in (await self._get_page_content(page)): + while "Just a moment..." in (await ResponseFactory._get_async_page_content(page)): log.info("Waiting for Cloudflare wait page to disappear.") await page.wait_for_timeout(1000) await page.wait_for_load_state() @@ -593,7 +563,7 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession): box_selector = "#cf_turnstile div, #cf-turnstile div, .turnstile>div>div" if challenge_type != "embedded": box_selector = ".main-content p+div>div>div" - while "Verifying you are human." in (await self._get_page_content(page)): + while "Verifying you are human." in (await ResponseFactory._get_async_page_content(page)): # Waiting for the verify spinner to disappear, checking every 1s if it disappeared await page.wait_for_timeout(500) @@ -753,7 +723,7 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession): # Create response object response = await ResponseFactory.from_async_playwright_response( - page_info.page, first_response, final_response, params.selector_config + page_info.page, first_response, final_response, params.selector_config, bool(params.page_action) ) # Close the page to free up resources diff --git a/scrapling/engines/_browsers/_controllers.py b/scrapling/engines/_browsers/_controllers.py index ca8b45e..45a6938 100644 --- a/scrapling/engines/_browsers/_controllers.py +++ b/scrapling/engines/_browsers/_controllers.py @@ -306,7 +306,7 @@ class DynamicSession(DynamicSessionMixin, SyncSession): # Create response object response = ResponseFactory.from_playwright_response( - page_info.page, first_response, final_response, params.selector_config + page_info.page, first_response, final_response, params.selector_config, bool(params.page_action) ) # Close the page to free up resources @@ -563,7 +563,7 @@ class AsyncDynamicSession(DynamicSessionMixin, AsyncSession): # Create response object response = await ResponseFactory.from_async_playwright_response( - page_info.page, first_response, final_response, params.selector_config + page_info.page, first_response, final_response, params.selector_config, bool(params.page_action) ) # Close the page to free up resources diff --git a/scrapling/engines/toolbelt/convertor.py b/scrapling/engines/toolbelt/convertor.py index b66b518..5bb4a49 100644 --- a/scrapling/engines/toolbelt/convertor.py +++ b/scrapling/engines/toolbelt/convertor.py @@ -2,6 +2,7 @@ from functools import lru_cache from re import compile as re_compile from curl_cffi.requests import Response as CurlResponse +from playwright._impl._errors import Error as PlaywrightError from playwright.sync_api import Page as SyncPage, Response as SyncResponse from playwright.async_api import Page as AsyncPage, Response as AsyncResponse @@ -84,6 +85,7 @@ class ResponseFactory: first_response: SyncResponse, final_response: Optional[SyncResponse], parser_arguments: Dict, + automated_page: bool = False, ) -> Response: """ Transforms a Playwright response into an internal `Response` object, encapsulating @@ -99,6 +101,7 @@ class ResponseFactory: :param first_response: An earlier or initial Playwright `Response` object that may serve as a fallback response in the absence of the final one. :param parser_arguments: A dictionary containing additional arguments needed for parsing or further customization of the returned `Response`. These arguments are dynamically unpacked into the `Response` object. + :param automated_page: If True, it means the `page_action` argument was being used, so the response retrieving method changes to use Playwright's page instead of the final response. :return: A fully populated `Response` object containing the page's URL, content, status, headers, cookies, and other derived metadata. :rtype: Response @@ -114,7 +117,7 @@ class ResponseFactory: history = cls._process_response_history(first_response, parser_arguments) try: - page_content = final_response.text() + page_content = final_response.text() if not automated_page else cls._get_page_content(page) except Exception as e: # pragma: no cover log.error(f"Error getting page content: {e}") page_content = "" @@ -179,6 +182,36 @@ class ResponseFactory: return history + @classmethod + def _get_page_content(cls, page: SyncPage) -> str: + """ + A workaround for the Playwright issue with `page.content()` on Windows. Ref.: https://github.com/microsoft/playwright/issues/16108 + :param page: The page to extract content from. + :return: + """ + while True: + try: + return page.content() or "" + except PlaywrightError: + page.wait_for_timeout(500) + continue + return "" # pyright: ignore + + @classmethod + async def _get_async_page_content(cls, page: AsyncPage) -> str: + """ + A workaround for the Playwright issue with `page.content()` on Windows. Ref.: https://github.com/microsoft/playwright/issues/16108 + :param page: The page to extract content from. + :return: + """ + while True: + try: + return (await page.content()) or "" + except PlaywrightError: + await page.wait_for_timeout(500) + continue + return "" # pyright: ignore + @classmethod async def from_async_playwright_response( cls, @@ -186,6 +219,7 @@ class ResponseFactory: first_response: AsyncResponse, final_response: Optional[AsyncResponse], parser_arguments: Dict, + automated_page: bool = False, ) -> Response: """ Transforms a Playwright response into an internal `Response` object, encapsulating @@ -201,6 +235,7 @@ class ResponseFactory: :param first_response: An earlier or initial Playwright `Response` object that may serve as a fallback response in the absence of the final one. :param parser_arguments: A dictionary containing additional arguments needed for parsing or further customization of the returned `Response`. These arguments are dynamically unpacked into the `Response` object. + :param automated_page: If True, it means the `page_action` argument was being used, so the response retrieving method changes to use Playwright's page instead of the final response. :return: A fully populated `Response` object containing the page's URL, content, status, headers, cookies, and other derived metadata. :rtype: Response @@ -216,7 +251,7 @@ class ResponseFactory: history = await cls._async_process_response_history(first_response, parser_arguments) try: - page_content = await final_response.text() + page_content = await (final_response.text() if not automated_page else cls._get_async_page_content(page)) except Exception as e: # pragma: no cover log.error(f"Error getting page content in async: {e}") page_content = ""