refactor(fetchers): Optimizing fetchers + making PlayWrightFetcher 10% faster
Check out the Discord server for full details
This commit is contained in:
+5
-153
@@ -19,7 +19,7 @@ from scrapling.core._types import (
|
||||
from scrapling.core.utils import log
|
||||
from scrapling.engines.toolbelt import (
|
||||
Response,
|
||||
StatusText,
|
||||
ResponseFactory,
|
||||
async_intercept_route,
|
||||
check_type_validity,
|
||||
construct_proxy_dict,
|
||||
@@ -143,92 +143,6 @@ class CamoufoxEngine:
|
||||
**self.additional_arguments,
|
||||
}
|
||||
|
||||
def _process_response_history(self, first_response):
|
||||
"""Process response history to build a list of Response objects"""
|
||||
history = []
|
||||
current_request = first_response.request.redirected_from
|
||||
|
||||
try:
|
||||
while current_request:
|
||||
try:
|
||||
current_response = current_request.response()
|
||||
history.insert(
|
||||
0,
|
||||
Response(
|
||||
url=current_request.url,
|
||||
# using current_response.text() will trigger "Error: Response.text: Response body is unavailable for redirect responses"
|
||||
text="",
|
||||
body=b"",
|
||||
status=current_response.status if current_response else 301,
|
||||
reason=(
|
||||
current_response.status_text
|
||||
or StatusText.get(current_response.status)
|
||||
)
|
||||
if current_response
|
||||
else StatusText.get(301),
|
||||
encoding=current_response.headers.get("content-type", "")
|
||||
or "utf-8",
|
||||
cookies=tuple(),
|
||||
headers=current_response.all_headers()
|
||||
if current_response
|
||||
else {},
|
||||
request_headers=current_request.all_headers(),
|
||||
**self.adaptor_arguments,
|
||||
),
|
||||
)
|
||||
except Exception as e:
|
||||
log.error(f"Error processing redirect: {e}")
|
||||
break
|
||||
|
||||
current_request = current_request.redirected_from
|
||||
except Exception as e:
|
||||
log.error(f"Error processing response history: {e}")
|
||||
|
||||
return history
|
||||
|
||||
async def _async_process_response_history(self, first_response):
|
||||
"""Process response history to build a list of Response objects"""
|
||||
history = []
|
||||
current_request = first_response.request.redirected_from
|
||||
|
||||
try:
|
||||
while current_request:
|
||||
try:
|
||||
current_response = await current_request.response()
|
||||
history.insert(
|
||||
0,
|
||||
Response(
|
||||
url=current_request.url,
|
||||
# using current_response.text() will trigger "Error: Response.text: Response body is unavailable for redirect responses"
|
||||
text="",
|
||||
body=b"",
|
||||
status=current_response.status if current_response else 301,
|
||||
reason=(
|
||||
current_response.status_text
|
||||
or StatusText.get(current_response.status)
|
||||
)
|
||||
if current_response
|
||||
else StatusText.get(301),
|
||||
encoding=current_response.headers.get("content-type", "")
|
||||
or "utf-8",
|
||||
cookies=tuple(),
|
||||
headers=await current_response.all_headers()
|
||||
if current_response
|
||||
else {},
|
||||
request_headers=await current_request.all_headers(),
|
||||
**self.adaptor_arguments,
|
||||
),
|
||||
)
|
||||
except Exception as e:
|
||||
log.error(f"Error processing redirect: {e}")
|
||||
break
|
||||
|
||||
current_request = current_request.redirected_from
|
||||
except Exception as e:
|
||||
log.error(f"Error processing response history: {e}")
|
||||
|
||||
return history
|
||||
|
||||
@staticmethod
|
||||
def __detect_cloudflare(page_content):
|
||||
"""
|
||||
@@ -429,39 +343,8 @@ class CamoufoxEngine:
|
||||
log.error(f"Error waiting for selector {self.wait_selector}: {e}")
|
||||
|
||||
page.wait_for_timeout(self.wait)
|
||||
# In case we didn't catch a document type somehow
|
||||
final_response = final_response if final_response else first_response
|
||||
if not final_response:
|
||||
raise ValueError("Failed to get a response from the page")
|
||||
|
||||
# This will be parsed inside `Response`
|
||||
encoding = (
|
||||
final_response.headers.get("content-type", "") or "utf-8"
|
||||
) # default encoding
|
||||
# PlayWright API sometimes give empty status text for some reason!
|
||||
status_text = final_response.status_text or StatusText.get(
|
||||
final_response.status
|
||||
)
|
||||
|
||||
history = self._process_response_history(first_response)
|
||||
try:
|
||||
page_content = page.content()
|
||||
except Exception as e:
|
||||
log.error(f"Error getting page content: {e}")
|
||||
page_content = ""
|
||||
|
||||
response = Response(
|
||||
url=page.url,
|
||||
text=page_content,
|
||||
body=page_content.encode("utf-8"),
|
||||
status=final_response.status,
|
||||
reason=status_text,
|
||||
encoding=encoding,
|
||||
cookies=tuple(dict(cookie) for cookie in page.context.cookies()),
|
||||
headers=first_response.all_headers(),
|
||||
request_headers=first_response.request.all_headers(),
|
||||
history=history,
|
||||
**self.adaptor_arguments,
|
||||
response = ResponseFactory.from_playwright_response(
|
||||
page, first_response, final_response, self.adaptor_arguments
|
||||
)
|
||||
page.close()
|
||||
context.close()
|
||||
@@ -534,39 +417,8 @@ class CamoufoxEngine:
|
||||
log.error(f"Error waiting for selector {self.wait_selector}: {e}")
|
||||
|
||||
await page.wait_for_timeout(self.wait)
|
||||
# In case we didn't catch a document type somehow
|
||||
final_response = final_response if final_response else first_response
|
||||
if not final_response:
|
||||
raise ValueError("Failed to get a response from the page")
|
||||
|
||||
# This will be parsed inside `Response`
|
||||
encoding = (
|
||||
final_response.headers.get("content-type", "") or "utf-8"
|
||||
) # default encoding
|
||||
# PlayWright API sometimes give empty status text for some reason!
|
||||
status_text = final_response.status_text or StatusText.get(
|
||||
final_response.status
|
||||
)
|
||||
|
||||
history = await self._async_process_response_history(first_response)
|
||||
try:
|
||||
page_content = await page.content()
|
||||
except Exception as e:
|
||||
log.error(f"Error getting page content in async: {e}")
|
||||
page_content = ""
|
||||
|
||||
response = Response(
|
||||
url=page.url,
|
||||
text=page_content,
|
||||
body=page_content.encode("utf-8"),
|
||||
status=final_response.status,
|
||||
reason=status_text,
|
||||
encoding=encoding,
|
||||
cookies=tuple(dict(cookie) for cookie in await page.context.cookies()),
|
||||
headers=await first_response.all_headers(),
|
||||
request_headers=await first_response.request.all_headers(),
|
||||
history=history,
|
||||
**self.adaptor_arguments,
|
||||
response = await ResponseFactory.from_async_playwright_response(
|
||||
page, first_response, final_response, self.adaptor_arguments
|
||||
)
|
||||
await page.close()
|
||||
await context.close()
|
||||
|
||||
+22
-165
@@ -1,5 +1,14 @@
|
||||
import json
|
||||
|
||||
from playwright.sync_api import sync_playwright
|
||||
from playwright.async_api import async_playwright
|
||||
from playwright.sync_api import Response as SyncPlaywrightResponse
|
||||
from playwright.async_api import Response as AsyncPlaywrightResponse
|
||||
from rebrowser_playwright.sync_api import sync_playwright as sync_rebrowser_playwright
|
||||
from rebrowser_playwright.async_api import (
|
||||
async_playwright as async_rebrowser_playwright,
|
||||
)
|
||||
|
||||
from scrapling.core._types import (
|
||||
Callable,
|
||||
Dict,
|
||||
@@ -12,7 +21,7 @@ from scrapling.core.utils import log, lru_cache
|
||||
from scrapling.engines.constants import DEFAULT_STEALTH_FLAGS, NSTBROWSER_DEFAULT_QUERY
|
||||
from scrapling.engines.toolbelt import (
|
||||
Response,
|
||||
StatusText,
|
||||
ResponseFactory,
|
||||
async_intercept_route,
|
||||
check_type_validity,
|
||||
construct_cdp_url,
|
||||
@@ -227,110 +236,22 @@ class PlaywrightEngine:
|
||||
)
|
||||
)
|
||||
|
||||
def _process_response_history(self, first_response):
|
||||
"""Process response history to build a list of Response objects"""
|
||||
history = []
|
||||
current_request = first_response.request.redirected_from
|
||||
|
||||
try:
|
||||
while current_request:
|
||||
try:
|
||||
current_response = current_request.response()
|
||||
history.insert(
|
||||
0,
|
||||
Response(
|
||||
url=current_request.url,
|
||||
# using current_response.text() will trigger "Error: Response.text: Response body is unavailable for redirect responses"
|
||||
text="",
|
||||
body=b"",
|
||||
status=current_response.status if current_response else 301,
|
||||
reason=(
|
||||
current_response.status_text
|
||||
or StatusText.get(current_response.status)
|
||||
)
|
||||
if current_response
|
||||
else StatusText.get(301),
|
||||
encoding=current_response.headers.get("content-type", "")
|
||||
or "utf-8",
|
||||
cookies=tuple(),
|
||||
headers=current_response.all_headers()
|
||||
if current_response
|
||||
else {},
|
||||
request_headers=current_request.all_headers(),
|
||||
**self.adaptor_arguments,
|
||||
),
|
||||
)
|
||||
except Exception as e:
|
||||
log.error(f"Error processing redirect: {e}")
|
||||
break
|
||||
|
||||
current_request = current_request.redirected_from
|
||||
except Exception as e:
|
||||
log.error(f"Error processing response history: {e}")
|
||||
|
||||
return history
|
||||
|
||||
async def _async_process_response_history(self, first_response):
|
||||
"""Process response history to build a list of Response objects"""
|
||||
history = []
|
||||
current_request = first_response.request.redirected_from
|
||||
|
||||
try:
|
||||
while current_request:
|
||||
try:
|
||||
current_response = await current_request.response()
|
||||
history.insert(
|
||||
0,
|
||||
Response(
|
||||
url=current_request.url,
|
||||
# using current_response.text() will trigger "Error: Response.text: Response body is unavailable for redirect responses"
|
||||
text="",
|
||||
body=b"",
|
||||
status=current_response.status if current_response else 301,
|
||||
reason=(
|
||||
current_response.status_text
|
||||
or StatusText.get(current_response.status)
|
||||
)
|
||||
if current_response
|
||||
else StatusText.get(301),
|
||||
encoding=current_response.headers.get("content-type", "")
|
||||
or "utf-8",
|
||||
cookies=tuple(),
|
||||
headers=await current_response.all_headers()
|
||||
if current_response
|
||||
else {},
|
||||
request_headers=await current_request.all_headers(),
|
||||
**self.adaptor_arguments,
|
||||
),
|
||||
)
|
||||
except Exception as e:
|
||||
log.error(f"Error processing redirect: {e}")
|
||||
break
|
||||
|
||||
current_request = current_request.redirected_from
|
||||
except Exception as e:
|
||||
log.error(f"Error processing response history: {e}")
|
||||
|
||||
return history
|
||||
|
||||
def fetch(self, url: str) -> Response:
|
||||
"""Opens up the browser and do your request based on your chosen options.
|
||||
|
||||
:param url: Target url.
|
||||
:return: A `Response` object that is the same as `Adaptor` object except it has these added attributes: `status`, `reason`, `cookies`, `headers`, and `request_headers`
|
||||
"""
|
||||
from playwright.sync_api import Response as PlaywrightResponse
|
||||
|
||||
sync_context = sync_rebrowser_playwright
|
||||
if not self.stealth or self.real_chrome:
|
||||
# Because rebrowser_playwright doesn't play well with real browsers
|
||||
from playwright.sync_api import sync_playwright
|
||||
else:
|
||||
from rebrowser_playwright.sync_api import sync_playwright
|
||||
sync_context = sync_playwright
|
||||
|
||||
final_response = None
|
||||
referer = generate_convincing_referer(url) if self.google_search else None
|
||||
|
||||
def handle_response(finished_response: PlaywrightResponse):
|
||||
def handle_response(finished_response: SyncPlaywrightResponse):
|
||||
nonlocal final_response
|
||||
if (
|
||||
finished_response.request.resource_type == "document"
|
||||
@@ -338,7 +259,7 @@ class PlaywrightEngine:
|
||||
):
|
||||
final_response = finished_response
|
||||
|
||||
with sync_playwright() as p:
|
||||
with sync_context() as p:
|
||||
# Creating the browser
|
||||
if self.cdp_url:
|
||||
cdp_url = self._cdp_url_logic()
|
||||
@@ -390,39 +311,8 @@ class PlaywrightEngine:
|
||||
log.error(f"Error waiting for selector {self.wait_selector}: {e}")
|
||||
|
||||
page.wait_for_timeout(self.wait)
|
||||
# In case we didn't catch a document type somehow
|
||||
final_response = final_response if final_response else first_response
|
||||
if not final_response:
|
||||
raise ValueError("Failed to get a response from the page")
|
||||
|
||||
# This will be parsed inside `Response`
|
||||
encoding = (
|
||||
final_response.headers.get("content-type", "") or "utf-8"
|
||||
) # default encoding
|
||||
# PlayWright API sometimes give empty status text for some reason!
|
||||
status_text = final_response.status_text or StatusText.get(
|
||||
final_response.status
|
||||
)
|
||||
|
||||
history = self._process_response_history(first_response)
|
||||
try:
|
||||
page_content = page.content()
|
||||
except Exception as e:
|
||||
log.error(f"Error getting page content: {e}")
|
||||
page_content = ""
|
||||
|
||||
response = Response(
|
||||
url=page.url,
|
||||
text=page_content,
|
||||
body=page_content.encode("utf-8"),
|
||||
status=final_response.status,
|
||||
reason=status_text,
|
||||
encoding=encoding,
|
||||
cookies=tuple(dict(cookie) for cookie in page.context.cookies()),
|
||||
headers=first_response.all_headers(),
|
||||
request_headers=first_response.request.all_headers(),
|
||||
history=history,
|
||||
**self.adaptor_arguments,
|
||||
response = ResponseFactory.from_playwright_response(
|
||||
page, first_response, final_response, self.adaptor_arguments
|
||||
)
|
||||
page.close()
|
||||
context.close()
|
||||
@@ -434,18 +324,16 @@ class PlaywrightEngine:
|
||||
:param url: Target url.
|
||||
:return: A `Response` object that is the same as `Adaptor` object except it has these added attributes: `status`, `reason`, `cookies`, `headers`, and `request_headers`
|
||||
"""
|
||||
from playwright.async_api import Response as PlaywrightResponse
|
||||
|
||||
async_context = async_rebrowser_playwright
|
||||
if not self.stealth or self.real_chrome:
|
||||
# Because rebrowser_playwright doesn't play well with real browsers
|
||||
from playwright.async_api import async_playwright
|
||||
else:
|
||||
from rebrowser_playwright.async_api import async_playwright
|
||||
async_context = async_playwright
|
||||
|
||||
final_response = None
|
||||
referer = generate_convincing_referer(url) if self.google_search else None
|
||||
|
||||
async def handle_response(finished_response: PlaywrightResponse):
|
||||
async def handle_response(finished_response: AsyncPlaywrightResponse):
|
||||
nonlocal final_response
|
||||
if (
|
||||
finished_response.request.resource_type == "document"
|
||||
@@ -453,7 +341,7 @@ class PlaywrightEngine:
|
||||
):
|
||||
final_response = finished_response
|
||||
|
||||
async with async_playwright() as p:
|
||||
async with async_context() as p:
|
||||
# Creating the browser
|
||||
if self.cdp_url:
|
||||
cdp_url = self._cdp_url_logic()
|
||||
@@ -505,39 +393,8 @@ class PlaywrightEngine:
|
||||
log.error(f"Error waiting for selector {self.wait_selector}: {e}")
|
||||
|
||||
await page.wait_for_timeout(self.wait)
|
||||
# In case we didn't catch a document type somehow
|
||||
final_response = final_response if final_response else first_response
|
||||
if not final_response:
|
||||
raise ValueError("Failed to get a response from the page")
|
||||
|
||||
# This will be parsed inside `Response`
|
||||
encoding = (
|
||||
final_response.headers.get("content-type", "") or "utf-8"
|
||||
) # default encoding
|
||||
# PlayWright API sometimes give empty status text for some reason!
|
||||
status_text = final_response.status_text or StatusText.get(
|
||||
final_response.status
|
||||
)
|
||||
|
||||
history = await self._async_process_response_history(first_response)
|
||||
try:
|
||||
page_content = await page.content()
|
||||
except Exception as e:
|
||||
log.error(f"Error getting page content in async: {e}")
|
||||
page_content = ""
|
||||
|
||||
response = Response(
|
||||
url=page.url,
|
||||
text=page_content,
|
||||
body=page_content.encode("utf-8"),
|
||||
status=final_response.status,
|
||||
reason=status_text,
|
||||
encoding=encoding,
|
||||
cookies=tuple(dict(cookie) for cookie in await page.context.cookies()),
|
||||
headers=await first_response.all_headers(),
|
||||
request_headers=await first_response.request.all_headers(),
|
||||
history=history,
|
||||
**self.adaptor_arguments,
|
||||
response = await ResponseFactory.from_async_playwright_response(
|
||||
page, first_response, final_response, self.adaptor_arguments
|
||||
)
|
||||
await page.close()
|
||||
await context.close()
|
||||
|
||||
@@ -14,3 +14,4 @@ from .navigation import (
|
||||
intercept_route,
|
||||
js_bypass_path,
|
||||
)
|
||||
from .convertor import ResponseFactory
|
||||
|
||||
@@ -0,0 +1,259 @@
|
||||
from curl_cffi.requests import Response as CurlResponse
|
||||
from playwright.sync_api import Page as SyncPage, Response as SyncResponse
|
||||
from playwright.async_api import Page as AsyncPage, Response as AsyncResponse
|
||||
|
||||
from scrapling.core.utils import log
|
||||
from scrapling.core._types import Dict, Optional
|
||||
from .custom import Response, StatusText
|
||||
|
||||
|
||||
class ResponseFactory:
|
||||
"""
|
||||
Factory class for creating `Response` objects from various sources.
|
||||
|
||||
This class provides multiple static and instance methods for building standardized `Response` objects
|
||||
from diverse input sources such as Playwright responses, asynchronous Playwright responses,
|
||||
and raw HTTP request responses. It supports handling response histories, constructing the proper
|
||||
response objects, and managing encoding, headers, cookies, and other attributes.
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def _process_response_history(
|
||||
cls, first_response: SyncResponse, parser_arguments: Dict
|
||||
) -> list[Response]:
|
||||
"""Process response history to build a list of `Response` objects"""
|
||||
history = []
|
||||
current_request = first_response.request.redirected_from
|
||||
|
||||
try:
|
||||
while current_request:
|
||||
try:
|
||||
current_response = current_request.response()
|
||||
history.insert(
|
||||
0,
|
||||
Response(
|
||||
url=current_request.url,
|
||||
# using current_response.text() will trigger "Error: Response.text: Response body is unavailable for redirect responses"
|
||||
text="",
|
||||
body=b"",
|
||||
status=current_response.status if current_response else 301,
|
||||
reason=(
|
||||
current_response.status_text
|
||||
or StatusText.get(current_response.status)
|
||||
)
|
||||
if current_response
|
||||
else StatusText.get(301),
|
||||
encoding=current_response.headers.get("content-type", "")
|
||||
or "utf-8",
|
||||
cookies=tuple(),
|
||||
headers=current_response.all_headers()
|
||||
if current_response
|
||||
else {},
|
||||
request_headers=current_request.all_headers(),
|
||||
**parser_arguments,
|
||||
),
|
||||
)
|
||||
except Exception as e:
|
||||
log.error(f"Error processing redirect: {e}")
|
||||
break
|
||||
|
||||
current_request = current_request.redirected_from
|
||||
except Exception as e:
|
||||
log.error(f"Error processing response history: {e}")
|
||||
|
||||
return history
|
||||
|
||||
@classmethod
|
||||
def from_playwright_response(
|
||||
cls,
|
||||
page: SyncPage,
|
||||
first_response: SyncResponse,
|
||||
final_response: Optional[SyncResponse],
|
||||
parser_arguments: Dict,
|
||||
) -> Response:
|
||||
"""
|
||||
Transforms a Playwright response into an internal `Response` object, encapsulating
|
||||
the page's content, response status, headers, and relevant metadata.
|
||||
|
||||
The function handles potential issues, such as empty or missing final responses,
|
||||
by falling back to the first response if necessary. Encoding and status text
|
||||
are also derived from the provided response headers or reasonable defaults.
|
||||
Additionally, the page content and cookies are extracted for further use.
|
||||
|
||||
:param page: A synchronous Playwright `Page` instance that represents the current browser page. Required to retrieve the page's URL, cookies, and content.
|
||||
:param final_response: The last response received for the given request from the Playwright instance. Typically used as the main response object to derive status, headers, and other metadata.
|
||||
:param first_response: An earlier or initial Playwright `Response` object that may serve as a fallback response in the absence of the final one.
|
||||
:param parser_arguments: A dictionary containing additional arguments needed for parsing or further customization of the returned `Response`. These arguments are dynamically unpacked into
|
||||
the `Response` object.
|
||||
|
||||
:return: A fully populated `Response` object containing the page's URL, content, status, headers, cookies, and other derived metadata.
|
||||
:rtype: Response
|
||||
"""
|
||||
# In case we didn't catch a document type somehow
|
||||
final_response = final_response if final_response else first_response
|
||||
if not final_response:
|
||||
raise ValueError("Failed to get a response from the page")
|
||||
|
||||
# This will be parsed inside `Response`
|
||||
encoding = (
|
||||
final_response.headers.get("content-type", "") or "utf-8"
|
||||
) # default encoding
|
||||
# PlayWright API sometimes give empty status text for some reason!
|
||||
status_text = final_response.status_text or StatusText.get(
|
||||
final_response.status
|
||||
)
|
||||
|
||||
history = cls._process_response_history(first_response, parser_arguments)
|
||||
try:
|
||||
page_content = page.content()
|
||||
except Exception as e:
|
||||
log.error(f"Error getting page content: {e}")
|
||||
page_content = ""
|
||||
|
||||
return Response(
|
||||
url=page.url,
|
||||
text=page_content,
|
||||
body=page_content.encode("utf-8"),
|
||||
status=final_response.status,
|
||||
reason=status_text,
|
||||
encoding=encoding,
|
||||
cookies=tuple(dict(cookie) for cookie in page.context.cookies()),
|
||||
headers=first_response.all_headers(),
|
||||
request_headers=first_response.request.all_headers(),
|
||||
history=history,
|
||||
**parser_arguments,
|
||||
)
|
||||
|
||||
@classmethod
|
||||
async def _async_process_response_history(
|
||||
cls, first_response: AsyncResponse, parser_arguments: Dict
|
||||
) -> list[Response]:
|
||||
"""Process response history to build a list of `Response` objects"""
|
||||
history = []
|
||||
current_request = first_response.request.redirected_from
|
||||
|
||||
try:
|
||||
while current_request:
|
||||
try:
|
||||
current_response = await current_request.response()
|
||||
history.insert(
|
||||
0,
|
||||
Response(
|
||||
url=current_request.url,
|
||||
# using current_response.text() will trigger "Error: Response.text: Response body is unavailable for redirect responses"
|
||||
text="",
|
||||
body=b"",
|
||||
status=current_response.status if current_response else 301,
|
||||
reason=(
|
||||
current_response.status_text
|
||||
or StatusText.get(current_response.status)
|
||||
)
|
||||
if current_response
|
||||
else StatusText.get(301),
|
||||
encoding=current_response.headers.get("content-type", "")
|
||||
or "utf-8",
|
||||
cookies=tuple(),
|
||||
headers=await current_response.all_headers()
|
||||
if current_response
|
||||
else {},
|
||||
request_headers=await current_request.all_headers(),
|
||||
**parser_arguments,
|
||||
),
|
||||
)
|
||||
except Exception as e:
|
||||
log.error(f"Error processing redirect: {e}")
|
||||
break
|
||||
|
||||
current_request = current_request.redirected_from
|
||||
except Exception as e:
|
||||
log.error(f"Error processing response history: {e}")
|
||||
|
||||
return history
|
||||
|
||||
@classmethod
|
||||
async def from_async_playwright_response(
|
||||
cls,
|
||||
page: AsyncPage,
|
||||
first_response: AsyncResponse,
|
||||
final_response: Optional[AsyncResponse],
|
||||
parser_arguments: Dict,
|
||||
) -> Response:
|
||||
"""
|
||||
Transforms a Playwright response into an internal `Response` object, encapsulating
|
||||
the page's content, response status, headers, and relevant metadata.
|
||||
|
||||
The function handles potential issues, such as empty or missing final responses,
|
||||
by falling back to the first response if necessary. Encoding and status text
|
||||
are also derived from the provided response headers or reasonable defaults.
|
||||
Additionally, the page content and cookies are extracted for further use.
|
||||
|
||||
:param page: An asynchronous Playwright `Page` instance that represents the current browser page. Required to retrieve the page's URL, cookies, and content.
|
||||
:param final_response: The last response received for the given request from the Playwright instance. Typically used as the main response object to derive status, headers, and other metadata.
|
||||
:param first_response: An earlier or initial Playwright `Response` object that may serve as a fallback response in the absence of the final one.
|
||||
:param parser_arguments: A dictionary containing additional arguments needed for parsing or further customization of the returned `Response`. These arguments are dynamically unpacked into
|
||||
the `Response` object.
|
||||
|
||||
:return: A fully populated `Response` object containing the page's URL, content, status, headers, cookies, and other derived metadata.
|
||||
:rtype: Response
|
||||
"""
|
||||
# In case we didn't catch a document type somehow
|
||||
final_response = final_response if final_response else first_response
|
||||
if not final_response:
|
||||
raise ValueError("Failed to get a response from the page")
|
||||
|
||||
# This will be parsed inside `Response`
|
||||
encoding = (
|
||||
final_response.headers.get("content-type", "") or "utf-8"
|
||||
) # default encoding
|
||||
# PlayWright API sometimes give empty status text for some reason!
|
||||
status_text = final_response.status_text or StatusText.get(
|
||||
final_response.status
|
||||
)
|
||||
|
||||
history = await cls._async_process_response_history(
|
||||
first_response, parser_arguments
|
||||
)
|
||||
try:
|
||||
page_content = await page.content()
|
||||
except Exception as e:
|
||||
log.error(f"Error getting page content in async: {e}")
|
||||
page_content = ""
|
||||
|
||||
return Response(
|
||||
url=page.url,
|
||||
text=page_content,
|
||||
body=page_content.encode("utf-8"),
|
||||
status=final_response.status,
|
||||
reason=status_text,
|
||||
encoding=encoding,
|
||||
cookies=tuple(dict(cookie) for cookie in await page.context.cookies()),
|
||||
headers=await first_response.all_headers(),
|
||||
request_headers=await first_response.request.all_headers(),
|
||||
history=history,
|
||||
**parser_arguments,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def from_http_request(response: CurlResponse, parser_arguments: Dict) -> Response:
|
||||
"""Takes `curl_cffi` response and generates `Response` object from it.
|
||||
|
||||
:param response: `curl_cffi` response object
|
||||
:param parser_arguments: Additional arguments to be passed to the `Response` object constructor.
|
||||
:return: A `Response` object that is the same as `Adaptor` object except it has these added attributes: `status`, `reason`, `cookies`, `headers`, and `request_headers`
|
||||
"""
|
||||
return Response(
|
||||
url=response.url,
|
||||
text=response.text,
|
||||
body=response.content
|
||||
if type(response.content) is bytes
|
||||
else response.content.encode(),
|
||||
status=response.status_code,
|
||||
reason=response.reason,
|
||||
encoding=response.encoding or "utf-8",
|
||||
cookies=dict(response.cookies),
|
||||
headers=dict(response.headers),
|
||||
request_headers=dict(response.request.headers),
|
||||
method=response.request.method,
|
||||
history=response.history, # https://github.com/lexiforest/curl_cffi/issues/82
|
||||
**parser_arguments,
|
||||
)
|
||||
Reference in New Issue
Block a user