fix(fetchers): More accurate collection of the final response

This will make the `response.body` return the raw response for text/json responses as well
This commit is contained in:
Karim shoair
2025-10-12 04:12:18 +03:00
parent 6bfa266c90
commit 073c4910d4
3 changed files with 11 additions and 11 deletions
+2
View File
@@ -344,6 +344,7 @@ class StealthySession(StealthySessionMixin, SyncSession):
if (
finished_response.request.resource_type == "document"
and finished_response.request.is_navigation_request()
and finished_response.request.frame == page_info.page.main_frame
):
final_response = finished_response
@@ -676,6 +677,7 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
if (
finished_response.request.resource_type == "document"
and finished_response.request.is_navigation_request()
and finished_response.request.frame == page_info.page.main_frame
):
final_response = finished_response
@@ -257,6 +257,7 @@ class DynamicSession(DynamicSessionMixin, SyncSession):
if (
finished_response.request.resource_type == "document"
and finished_response.request.is_navigation_request()
and finished_response.request.frame == page_info.page.main_frame
):
final_response = finished_response
@@ -503,6 +504,7 @@ class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
if (
finished_response.request.resource_type == "document"
and finished_response.request.is_navigation_request()
and finished_response.request.frame == page_info.page.main_frame
):
final_response = finished_response
+7 -11
View File
@@ -24,15 +24,15 @@ class ResponseFactory:
@classmethod
@lru_cache(maxsize=16)
def __extract_browser_encoding(cls, content_type: str | None) -> Optional[str]:
def __extract_browser_encoding(cls, content_type: str | None, default: str = "utf-8") -> str:
"""Extract browser encoding from headers.
Ex: from header "content-type: text/html; charset=utf-8" -> "utf-8
"""
if content_type:
# Because Playwright can't do that by themselves like all libraries for some reason :3
match = __CHARSET_RE__.search(content_type)
return match.group(1) if match else None
return None
return match.group(1) if match else default
return default
@classmethod
def _process_response_history(cls, first_response: SyncResponse, parser_arguments: Dict) -> list[Response]:
@@ -108,15 +108,13 @@ class ResponseFactory:
if not final_response:
raise ValueError("Failed to get a response from the page")
encoding = (
cls.__extract_browser_encoding(final_response.headers.get("content-type", "")) or "utf-8"
) # default encoding
encoding = cls.__extract_browser_encoding(final_response.headers.get("content-type", ""))
# PlayWright API sometimes give empty status text for some reason!
status_text = final_response.status_text or StatusText.get(final_response.status)
history = cls._process_response_history(first_response, parser_arguments)
try:
page_content = page.content()
page_content = final_response.text()
except Exception as e: # pragma: no cover
log.error(f"Error getting page content: {e}")
page_content = ""
@@ -212,15 +210,13 @@ class ResponseFactory:
if not final_response:
raise ValueError("Failed to get a response from the page")
encoding = (
cls.__extract_browser_encoding(final_response.headers.get("content-type", "")) or "utf-8"
) # default encoding
encoding = cls.__extract_browser_encoding(final_response.headers.get("content-type", ""))
# PlayWright API sometimes give empty status text for some reason!
status_text = final_response.status_text or StatusText.get(final_response.status)
history = await cls._async_process_response_history(first_response, parser_arguments)
try:
page_content = await page.content()
page_content = await final_response.text()
except Exception as e: # pragma: no cover
log.error(f"Error getting page content in async: {e}")
page_content = ""