fix: complete MCP schema validation for remaining bare array types

- Fix params (Dict|List|Tuple → Dict), urls (Tuple[str,...] → List[str]),
- Renamed _NormalizeCredentials/_ContentTranslator to snake_case.
- Also, raise a ValueError on invalid credentials instead of returning None silently.
This commit is contained in:
Karim shoair
2026-02-27 00:28:44 +02:00
parent 86e0f0f250
commit 921314c314
+21 -21
View File
@@ -37,12 +37,12 @@ class ResponseModel(BaseModel):
url: str = Field(description="The URL given by the user that resulted in this response.") url: str = Field(description="The URL given by the user that resulted in this response.")
def _ContentTranslator(content: Generator[str, None, None], page: _ScraplingResponse) -> ResponseModel: def _content_translator(content: Generator[str, None, None], page: _ScraplingResponse) -> ResponseModel:
"""Convert a content generator to a list of ResponseModel objects.""" """Convert a content generator to a list of ResponseModel objects."""
return ResponseModel(status=page.status, content=[result for result in content], url=page.url) return ResponseModel(status=page.status, content=[result for result in content], url=page.url)
def _NormalizeCredentials(credentials: Optional[Dict[str, str]]) -> Optional[Tuple[str, str]]: def _normalize_credentials(credentials: Optional[Dict[str, str]]) -> Optional[Tuple[str, str]]:
"""Convert a credentials dictionary to a tuple accepted by fetchers.""" """Convert a credentials dictionary to a tuple accepted by fetchers."""
if not credentials: if not credentials:
return None return None
@@ -51,7 +51,7 @@ def _NormalizeCredentials(credentials: Optional[Dict[str, str]]) -> Optional[Tup
password = credentials.get("password") password = credentials.get("password")
if username is None or password is None: if username is None or password is None:
return None raise ValueError("Credentials dictionary must contain both 'username' and 'password' keys")
return username, password return username, password
@@ -64,7 +64,7 @@ class ScraplingMCPServer:
extraction_type: extraction_types = "markdown", extraction_type: extraction_types = "markdown",
css_selector: Optional[str] = None, css_selector: Optional[str] = None,
main_content_only: bool = True, main_content_only: bool = True,
params: Optional[Dict | List | Tuple] = None, params: Optional[Dict] = None,
headers: Optional[Mapping[str, Optional[str]]] = None, headers: Optional[Mapping[str, Optional[str]]] = None,
cookies: Optional[Dict[str, str]] = None, cookies: Optional[Dict[str, str]] = None,
timeout: Optional[int | float] = 30, timeout: Optional[int | float] = 30,
@@ -107,8 +107,8 @@ class ScraplingMCPServer:
:param http3: Whether to use HTTP3. Defaults to False. It might be problematic if used it with `impersonate`. :param http3: Whether to use HTTP3. Defaults to False. It might be problematic if used it with `impersonate`.
:param stealthy_headers: If enabled (default), it creates and adds real browser headers. It also sets the referer header as if this request came from a Google search of URL's domain. :param stealthy_headers: If enabled (default), it creates and adds real browser headers. It also sets the referer header as if this request came from a Google search of URL's domain.
""" """
normalized_proxy_auth = _NormalizeCredentials(proxy_auth) normalized_proxy_auth = _normalize_credentials(proxy_auth)
normalized_auth = _NormalizeCredentials(auth) normalized_auth = _normalize_credentials(auth)
page = Fetcher.get( page = Fetcher.get(
url, url,
@@ -128,7 +128,7 @@ class ScraplingMCPServer:
max_redirects=max_redirects, max_redirects=max_redirects,
follow_redirects=follow_redirects, follow_redirects=follow_redirects,
) )
return _ContentTranslator( return _content_translator(
Convertor._extract_content( Convertor._extract_content(
page, page,
css_selector=css_selector, css_selector=css_selector,
@@ -140,12 +140,12 @@ class ScraplingMCPServer:
@staticmethod @staticmethod
async def bulk_get( async def bulk_get(
urls: Tuple[str, ...], urls: List[str],
impersonate: ImpersonateType = "chrome", impersonate: ImpersonateType = "chrome",
extraction_type: extraction_types = "markdown", extraction_type: extraction_types = "markdown",
css_selector: Optional[str] = None, css_selector: Optional[str] = None,
main_content_only: bool = True, main_content_only: bool = True,
params: Optional[Dict | List | Tuple] = None, params: Optional[Dict] = None,
headers: Optional[Mapping[str, Optional[str]]] = None, headers: Optional[Mapping[str, Optional[str]]] = None,
cookies: Optional[Dict[str, str]] = None, cookies: Optional[Dict[str, str]] = None,
timeout: Optional[int | float] = 30, timeout: Optional[int | float] = 30,
@@ -164,7 +164,7 @@ class ScraplingMCPServer:
Note: This is only suitable for low-mid protection levels. For high-protection levels or websites that require JS loading, use the other tools directly. Note: This is only suitable for low-mid protection levels. For high-protection levels or websites that require JS loading, use the other tools directly.
Note: If the `css_selector` resolves to more than one element, all the elements will be returned. Note: If the `css_selector` resolves to more than one element, all the elements will be returned.
:param urls: A tuple of the URLs to request. :param urls: A list of the URLs to request.
:param impersonate: Browser version to impersonate its fingerprint. It's using the latest chrome version by default. :param impersonate: Browser version to impersonate its fingerprint. It's using the latest chrome version by default.
:param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are: :param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are:
- Markdown will convert the page content to Markdown format. - Markdown will convert the page content to Markdown format.
@@ -188,8 +188,8 @@ class ScraplingMCPServer:
:param http3: Whether to use HTTP3. Defaults to False. It might be problematic if used it with `impersonate`. :param http3: Whether to use HTTP3. Defaults to False. It might be problematic if used it with `impersonate`.
:param stealthy_headers: If enabled (default), it creates and adds real browser headers. It also sets the referer header as if this request came from a Google search of URL's domain. :param stealthy_headers: If enabled (default), it creates and adds real browser headers. It also sets the referer header as if this request came from a Google search of URL's domain.
""" """
normalized_proxy_auth = _NormalizeCredentials(proxy_auth) normalized_proxy_auth = _normalize_credentials(proxy_auth)
normalized_auth = _NormalizeCredentials(auth) normalized_auth = _normalize_credentials(auth)
async with FetcherSession() as session: async with FetcherSession() as session:
tasks: List[Any] = [ tasks: List[Any] = [
@@ -215,7 +215,7 @@ class ScraplingMCPServer:
] ]
responses = await gather(*tasks) responses = await gather(*tasks)
return [ return [
_ContentTranslator( _content_translator(
Convertor._extract_content( Convertor._extract_content(
page, page,
css_selector=css_selector, css_selector=css_selector,
@@ -299,7 +299,7 @@ class ScraplingMCPServer:
disable_resources=disable_resources, disable_resources=disable_resources,
wait_selector_state=wait_selector_state, wait_selector_state=wait_selector_state,
) )
return _ContentTranslator( return _content_translator(
Convertor._extract_content( Convertor._extract_content(
page, page,
css_selector=css_selector, css_selector=css_selector,
@@ -311,7 +311,7 @@ class ScraplingMCPServer:
@staticmethod @staticmethod
async def bulk_fetch( async def bulk_fetch(
urls: Tuple[str, ...], urls: List[str],
extraction_type: extraction_types = "markdown", extraction_type: extraction_types = "markdown",
css_selector: Optional[str] = None, css_selector: Optional[str] = None,
main_content_only: bool = True, main_content_only: bool = True,
@@ -336,7 +336,7 @@ class ScraplingMCPServer:
Note: This is only suitable for low-mid protection levels. Note: This is only suitable for low-mid protection levels.
Note: If the `css_selector` resolves to more than one element, all the elements will be returned. Note: If the `css_selector` resolves to more than one element, all the elements will be returned.
:param urls: A tuple of the URLs to request. :param urls: A list of the URLs to request.
:param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are: :param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are:
- Markdown will convert the page content to Markdown format. - Markdown will convert the page content to Markdown format.
- HTML will return the raw HTML content of the page. - HTML will return the raw HTML content of the page.
@@ -384,7 +384,7 @@ class ScraplingMCPServer:
tasks = [session.fetch(url) for url in urls] tasks = [session.fetch(url) for url in urls]
responses = await gather(*tasks) responses = await gather(*tasks)
return [ return [
_ContentTranslator( _content_translator(
Convertor._extract_content( Convertor._extract_content(
page, page,
css_selector=css_selector, css_selector=css_selector,
@@ -483,7 +483,7 @@ class ScraplingMCPServer:
disable_resources=disable_resources, disable_resources=disable_resources,
wait_selector_state=wait_selector_state, wait_selector_state=wait_selector_state,
) )
return _ContentTranslator( return _content_translator(
Convertor._extract_content( Convertor._extract_content(
page, page,
css_selector=css_selector, css_selector=css_selector,
@@ -495,7 +495,7 @@ class ScraplingMCPServer:
@staticmethod @staticmethod
async def bulk_stealthy_fetch( async def bulk_stealthy_fetch(
urls: Tuple[str, ...], urls: List[str],
extraction_type: extraction_types = "markdown", extraction_type: extraction_types = "markdown",
css_selector: Optional[str] = None, css_selector: Optional[str] = None,
main_content_only: bool = True, main_content_only: bool = True,
@@ -525,7 +525,7 @@ class ScraplingMCPServer:
Note: This is the only suitable fetcher for high protection levels. Note: This is the only suitable fetcher for high protection levels.
Note: If the `css_selector` resolves to more than one element, all the elements will be returned. Note: If the `css_selector` resolves to more than one element, all the elements will be returned.
:param urls: A tuple of the URLs to request. :param urls: A list of the URLs to request.
:param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are: :param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are:
- Markdown will convert the page content to Markdown format. - Markdown will convert the page content to Markdown format.
- HTML will return the raw HTML content of the page. - HTML will return the raw HTML content of the page.
@@ -582,7 +582,7 @@ class ScraplingMCPServer:
tasks = [session.fetch(url) for url in urls] tasks = [session.fetch(url) for url in urls]
responses = await gather(*tasks) responses = await gather(*tasks)
return [ return [
_ContentTranslator( _content_translator(
Convertor._extract_content( Convertor._extract_content(
page, page,
css_selector=css_selector, css_selector=css_selector,