feat(mcp): Add three new tools to control browser sessions

Now you can open a browser, keep using it for other requests as you want, and close it when you want.
This commit is contained in:
Karim shoair
2026-03-28 00:16:14 +02:00
parent a356dd2f2b
commit c458ab65a2
2 changed files with 441 additions and 54 deletions
+321 -53
View File
@@ -1,4 +1,7 @@
from uuid import uuid4
from asyncio import gather from asyncio import gather
from datetime import datetime, timezone
from dataclasses import dataclass, field
from mcp.server.fastmcp import FastMCP from mcp.server.fastmcp import FastMCP
from pydantic import BaseModel, Field from pydantic import BaseModel, Field
@@ -13,6 +16,7 @@ from scrapling.fetchers import (
) )
from scrapling.core._types import ( from scrapling.core._types import (
Optional, Optional,
Literal,
Tuple, Tuple,
Mapping, Mapping,
Dict, Dict,
@@ -24,6 +28,8 @@ from scrapling.core._types import (
SelectorWaitStates, SelectorWaitStates,
) )
SessionType = Literal["dynamic", "stealthy"]
class ResponseModel(BaseModel): class ResponseModel(BaseModel):
"""Request's response information structure.""" """Request's response information structure."""
@@ -33,6 +39,35 @@ class ResponseModel(BaseModel):
url: str = Field(description="The URL given by the user that resulted in this response.") url: str = Field(description="The URL given by the user that resulted in this response.")
class SessionInfo(BaseModel):
"""Information about an open browser session."""
session_id: str = Field(description="The unique identifier of the session.")
session_type: SessionType = Field(description="The type of the session: 'dynamic' or 'stealthy'.")
created_at: str = Field(description="ISO timestamp of when the session was created.")
is_alive: bool = Field(description="Whether the session is still alive and usable.")
class SessionCreatedModel(SessionInfo):
"""Response returned when a new session is created."""
message: str = Field(description="A confirmation message.")
class SessionClosedModel(BaseModel):
"""Response returned when a session is closed."""
session_id: str = Field(description="The unique identifier of the closed session.")
message: str = Field(description="A confirmation message.")
@dataclass
class _SessionEntry:
session: Any # AsyncDynamicSession | AsyncStealthySession
session_type: SessionType
created_at: str = field(default_factory=lambda: datetime.now(timezone.utc).isoformat())
def _translate_response( def _translate_response(
page: _ScraplingResponse, page: _ScraplingResponse,
extraction_type: extraction_types, extraction_type: extraction_types,
@@ -65,7 +100,177 @@ def _normalize_credentials(credentials: Optional[Dict[str, str]]) -> Optional[Tu
return username, password return username, password
def _build_fetch_kwargs(**params) -> Dict[str, Any]:
"""Build kwargs dict for session.fetch() from request-level parameters."""
kwargs: Dict[str, Any] = {}
# These are the params that session.fetch() accepts as per-request overrides
request_level_keys = (
"wait",
"timeout",
"google_search",
"extra_headers",
"disable_resources",
"wait_selector",
"wait_selector_state",
"network_idle",
"proxy",
"solve_cloudflare",
)
for key in request_level_keys:
if key in params and params[key] is not None:
kwargs[key] = params[key]
return kwargs
class ScraplingMCPServer: class ScraplingMCPServer:
def __init__(self):
self._sessions: Dict[str, _SessionEntry] = {}
def _get_session(self, session_id: str, expected_type: SessionType) -> _SessionEntry:
"""Look up a session by ID and validate its type."""
entry = self._sessions.get(session_id)
if entry is None:
raise ValueError(f"Session '{session_id}' not found. Use list_sessions to see active sessions.")
if not entry.session._is_alive:
raise ValueError(f"Session '{session_id}' is no longer alive. Open a new session.")
if entry.session_type != expected_type:
raise ValueError(
f"Session '{session_id}' is a '{entry.session_type}' session, but this tool requires a "
f"'{expected_type}' session. Use the matching fetch tool for your session type."
)
return entry
async def open_session(
self,
session_type: SessionType,
headless: bool = True,
google_search: bool = True,
real_chrome: bool = False,
wait: int | float = 0,
proxy: Optional[str | Dict[str, str]] = None,
timezone_id: str | None = None,
locale: str | None = None,
extra_headers: Optional[Dict[str, str]] = None,
useragent: Optional[str] = None,
cdp_url: Optional[str] = None,
timeout: int | float = 30000,
disable_resources: bool = False,
wait_selector: Optional[str] = None,
cookies: Sequence[SetCookieParam] | None = None,
network_idle: bool = False,
wait_selector_state: SelectorWaitStates = "attached",
max_pages: int = 5,
# Stealthy-only params (ignored for dynamic sessions)
hide_canvas: bool = False,
block_webrtc: bool = False,
allow_webgl: bool = True,
solve_cloudflare: bool = False,
additional_args: Optional[Dict] = None,
) -> SessionCreatedModel:
"""Open a persistent browser session that can be reused across multiple fetch calls.
This avoids the overhead of launching a new browser for each request.
Use close_session to close the session when done, and list_sessions to see all active sessions.
:param session_type: The type of session to open. Use "dynamic" for standard Playwright browser, or "stealthy" for anti-bot bypass with fingerprint spoofing.
:param headless: Run the browser in headless/hidden (default), or headful/visible mode.
:param google_search: Enabled by default, Scrapling will set a Google referer header.
:param real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it.
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the Response object.
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
:param timezone_id: Changes the timezone of the browser. Defaults to the system timezone.
:param locale: Specify user locale, for example, `en-GB`, `de-DE`, etc.
:param extra_headers: A dictionary of extra headers to add to the request.
:param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it.
:param cdp_url: Instead of launching a new browser instance, connect to this CDP URL to control real browsers through CDP.
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000.
:param disable_resources: Drop requests for unnecessary resources for a speed boost.
:param wait_selector: Wait for a specific CSS selector to be in a specific state.
:param cookies: Set cookies for the session. It should be in a dictionary format that Playwright accepts.
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
:param max_pages: Maximum number of concurrent pages/tabs in the browser. Defaults to 5. Higher values allow more parallel fetches.
:param hide_canvas: (Stealthy only) Add random noise to canvas operations to prevent fingerprinting.
:param block_webrtc: (Stealthy only) Forces WebRTC to respect proxy settings to prevent local IP address leak.
:param allow_webgl: (Stealthy only) Enabled by default. Disabling WebGL is not recommended as many WAFs now check if WebGL is enabled.
:param solve_cloudflare: (Stealthy only) Solves all types of the Cloudflare's Turnstile/Interstitial challenges.
:param additional_args: (Stealthy only) Additional arguments to be passed to Playwright's context as additional settings.
"""
common_kwargs: Dict[str, Any] = dict(
wait=wait,
proxy=proxy,
locale=locale,
timeout=timeout,
cookies=cookies,
cdp_url=cdp_url,
headless=headless,
max_pages=max_pages,
useragent=useragent,
timezone_id=timezone_id,
real_chrome=real_chrome,
network_idle=network_idle,
wait_selector=wait_selector,
google_search=google_search,
extra_headers=extra_headers,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
)
if session_type == "stealthy":
session = AsyncStealthySession(
**common_kwargs,
hide_canvas=hide_canvas,
block_webrtc=block_webrtc,
allow_webgl=allow_webgl,
solve_cloudflare=solve_cloudflare,
additional_args=additional_args,
)
else:
session = AsyncDynamicSession(**common_kwargs)
await session.start()
session_id = uuid4().hex[:12]
entry = _SessionEntry(session=session, session_type=session_type)
self._sessions[session_id] = entry
return SessionCreatedModel(
session_id=session_id,
session_type=session_type,
created_at=entry.created_at,
is_alive=True,
message=f"Session '{session_id}' ({session_type}) created successfully.",
)
async def close_session(
self,
session_id: str,
) -> SessionClosedModel:
"""Close a persistent browser session and free its resources.
:param session_id: The unique identifier of the session to close. Use list_sessions to see active sessions.
"""
entry = self._sessions.pop(session_id, None)
if entry is None:
raise ValueError(f"Session '{session_id}' not found. Use list_sessions to see active sessions.")
await entry.session.close()
return SessionClosedModel(
session_id=session_id,
message=f"Session '{session_id}' closed successfully.",
)
async def list_sessions(self) -> List[SessionInfo]:
"""List all active browser sessions with their details."""
return [
SessionInfo(
session_id=sid,
session_type=entry.session_type,
created_at=entry.created_at,
is_alive=entry.session._is_alive,
)
for sid, entry in self._sessions.items()
]
@staticmethod @staticmethod
async def get( async def get(
url: str, url: str,
@@ -217,8 +422,8 @@ class ScraplingMCPServer:
responses = await gather(*tasks) responses = await gather(*tasks)
return [_translate_response(page, extraction_type, css_selector, main_content_only) for page in responses] return [_translate_response(page, extraction_type, css_selector, main_content_only) for page in responses]
@staticmethod
async def fetch( async def fetch(
self,
url: str, url: str,
extraction_type: extraction_types = "markdown", extraction_type: extraction_types = "markdown",
css_selector: Optional[str] = None, css_selector: Optional[str] = None,
@@ -239,10 +444,13 @@ class ScraplingMCPServer:
cookies: Sequence[SetCookieParam] | None = None, cookies: Sequence[SetCookieParam] | None = None,
network_idle: bool = False, network_idle: bool = False,
wait_selector_state: SelectorWaitStates = "attached", wait_selector_state: SelectorWaitStates = "attached",
session_id: Optional[str] = None,
) -> ResponseModel: ) -> ResponseModel:
"""Use playwright to open a browser to fetch a URL and return a structured output of the result. """Use playwright to open a browser to fetch a URL and return a structured output of the result.
Note: This is only suitable for low-mid protection levels. Note: This is only suitable for low-mid protection levels.
Note: If the `css_selector` resolves to more than one element, all the elements will be returned. Note: If the `css_selector` resolves to more than one element, all the elements will be returned.
Note: If a `session_id` is provided (from open_session), the browser session will be reused instead of creating a new one.
When using a session, browser-level params (headless, proxy, locale, etc.) are ignored — they were set at session creation time.
:param url: The URL to request. :param url: The URL to request.
:param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are: :param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are:
@@ -269,8 +477,9 @@ class ScraplingMCPServer:
:param google_search: Enabled by default, Scrapling will set a Google referer header. :param google_search: Enabled by default, Scrapling will set a Google referer header.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by `google_search` takes priority over the referer set here if used together._ :param extra_headers: A dictionary of extra headers to add to the request. _The referer set by `google_search` takes priority over the referer set here if used together._
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. :param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
:param session_id: Optional session ID from open_session. If provided, reuses the existing browser session instead of creating a new one.
""" """
results = await ScraplingMCPServer.bulk_fetch( results = await self.bulk_fetch(
urls=[url], urls=[url],
extraction_type=extraction_type, extraction_type=extraction_type,
css_selector=css_selector, css_selector=css_selector,
@@ -291,11 +500,12 @@ class ScraplingMCPServer:
cookies=cookies, cookies=cookies,
network_idle=network_idle, network_idle=network_idle,
wait_selector_state=wait_selector_state, wait_selector_state=wait_selector_state,
session_id=session_id,
) )
return results[0] return results[0]
@staticmethod
async def bulk_fetch( async def bulk_fetch(
self,
urls: List[str], urls: List[str],
extraction_type: extraction_types = "markdown", extraction_type: extraction_types = "markdown",
css_selector: Optional[str] = None, css_selector: Optional[str] = None,
@@ -316,10 +526,13 @@ class ScraplingMCPServer:
cookies: Sequence[SetCookieParam] | None = None, cookies: Sequence[SetCookieParam] | None = None,
network_idle: bool = False, network_idle: bool = False,
wait_selector_state: SelectorWaitStates = "attached", wait_selector_state: SelectorWaitStates = "attached",
session_id: Optional[str] = None,
) -> List[ResponseModel]: ) -> List[ResponseModel]:
"""Use playwright to open a browser, then fetch a group of URLs at the same time, and for each page return a structured output of the result. """Use playwright to open a browser, then fetch a group of URLs at the same time, and for each page return a structured output of the result.
Note: This is only suitable for low-mid protection levels. Note: This is only suitable for low-mid protection levels.
Note: If the `css_selector` resolves to more than one element, all the elements will be returned. Note: If the `css_selector` resolves to more than one element, all the elements will be returned.
Note: If a `session_id` is provided (from open_session), the browser session will be reused instead of creating a new one.
When using a session, browser-level params (headless, proxy, locale, etc.) are ignored — they were set at session creation time.
:param urls: A list of the URLs to request. :param urls: A list of the URLs to request.
:param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are: :param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are:
@@ -346,32 +559,50 @@ class ScraplingMCPServer:
:param google_search: Enabled by default, Scrapling will set a Google referer header. :param google_search: Enabled by default, Scrapling will set a Google referer header.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by `google_search` takes priority over the referer set here if used together._ :param extra_headers: A dictionary of extra headers to add to the request. _The referer set by `google_search` takes priority over the referer set here if used together._
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. :param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
:param session_id: Optional session ID from open_session. If provided, reuses the existing browser session instead of creating a new one.
""" """
async with AsyncDynamicSession( if session_id:
wait=wait, entry = self._get_session(session_id, "dynamic")
proxy=proxy, fetch_kwargs = _build_fetch_kwargs(
locale=locale, wait=wait,
timeout=timeout, timeout=timeout,
cookies=cookies, google_search=google_search,
cdp_url=cdp_url, extra_headers=extra_headers,
headless=headless, disable_resources=disable_resources,
max_pages=len(urls), wait_selector=wait_selector,
useragent=useragent, wait_selector_state=wait_selector_state,
timezone_id=timezone_id, network_idle=network_idle,
real_chrome=real_chrome, proxy=proxy,
network_idle=network_idle, )
wait_selector=wait_selector, tasks = [entry.session.fetch(url, **fetch_kwargs) for url in urls]
google_search=google_search,
extra_headers=extra_headers,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
) as session:
tasks = [session.fetch(url) for url in urls]
responses = await gather(*tasks) responses = await gather(*tasks)
return [_translate_response(page, extraction_type, css_selector, main_content_only) for page in responses] else:
async with AsyncDynamicSession(
wait=wait,
proxy=proxy,
locale=locale,
timeout=timeout,
cookies=cookies,
cdp_url=cdp_url,
headless=headless,
max_pages=len(urls),
useragent=useragent,
timezone_id=timezone_id,
real_chrome=real_chrome,
network_idle=network_idle,
wait_selector=wait_selector,
google_search=google_search,
extra_headers=extra_headers,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
) as session:
tasks = [session.fetch(url) for url in urls]
responses = await gather(*tasks)
return [_translate_response(page, extraction_type, css_selector, main_content_only) for page in responses]
@staticmethod
async def stealthy_fetch( async def stealthy_fetch(
self,
url: str, url: str,
extraction_type: extraction_types = "markdown", extraction_type: extraction_types = "markdown",
css_selector: Optional[str] = None, css_selector: Optional[str] = None,
@@ -397,10 +628,13 @@ class ScraplingMCPServer:
allow_webgl: bool = True, allow_webgl: bool = True,
solve_cloudflare: bool = False, solve_cloudflare: bool = False,
additional_args: Optional[Dict] = None, additional_args: Optional[Dict] = None,
session_id: Optional[str] = None,
) -> ResponseModel: ) -> ResponseModel:
"""Use the stealthy fetcher to fetch a URL and return a structured output of the result. """Use the stealthy fetcher to fetch a URL and return a structured output of the result.
Note: This is the only suitable fetcher for high protection levels. Note: This is the only suitable fetcher for high protection levels.
Note: If the `css_selector` resolves to more than one element, all the elements will be returned. Note: If the `css_selector` resolves to more than one element, all the elements will be returned.
Note: If a `session_id` is provided (from open_session), the browser session will be reused instead of creating a new one.
When using a session, browser-level params (headless, proxy, locale, etc.) are ignored — they were set at session creation time.
:param url: The URL to request. :param url: The URL to request.
:param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are: :param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are:
@@ -432,8 +666,9 @@ class ScraplingMCPServer:
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by `google_search` takes priority over the referer set here if used together._ :param extra_headers: A dictionary of extra headers to add to the request. _The referer set by `google_search` takes priority over the referer set here if used together._
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. :param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
:param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings. :param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings.
:param session_id: Optional session ID from open_session. If provided, reuses the existing browser session instead of creating a new one.
""" """
results = await ScraplingMCPServer.bulk_stealthy_fetch( results = await self.bulk_stealthy_fetch(
urls=[url], urls=[url],
extraction_type=extraction_type, extraction_type=extraction_type,
css_selector=css_selector, css_selector=css_selector,
@@ -459,11 +694,12 @@ class ScraplingMCPServer:
allow_webgl=allow_webgl, allow_webgl=allow_webgl,
solve_cloudflare=solve_cloudflare, solve_cloudflare=solve_cloudflare,
additional_args=additional_args, additional_args=additional_args,
session_id=session_id,
) )
return results[0] return results[0]
@staticmethod
async def bulk_stealthy_fetch( async def bulk_stealthy_fetch(
self,
urls: List[str], urls: List[str],
extraction_type: extraction_types = "markdown", extraction_type: extraction_types = "markdown",
css_selector: Optional[str] = None, css_selector: Optional[str] = None,
@@ -489,10 +725,13 @@ class ScraplingMCPServer:
allow_webgl: bool = True, allow_webgl: bool = True,
solve_cloudflare: bool = False, solve_cloudflare: bool = False,
additional_args: Optional[Dict] = None, additional_args: Optional[Dict] = None,
session_id: Optional[str] = None,
) -> List[ResponseModel]: ) -> List[ResponseModel]:
"""Use the stealthy fetcher to fetch a group of URLs at the same time, and for each page return a structured output of the result. """Use the stealthy fetcher to fetch a group of URLs at the same time, and for each page return a structured output of the result.
Note: This is the only suitable fetcher for high protection levels. Note: This is the only suitable fetcher for high protection levels.
Note: If the `css_selector` resolves to more than one element, all the elements will be returned. Note: If the `css_selector` resolves to more than one element, all the elements will be returned.
Note: If a `session_id` is provided (from open_session), the browser session will be reused instead of creating a new one.
When using a session, browser-level params (headless, proxy, locale, etc.) are ignored — they were set at session creation time.
:param urls: A list of the URLs to request. :param urls: A list of the URLs to request.
:param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are: :param extraction_type: The type of content to extract from the page. Defaults to "markdown". Options are:
@@ -524,45 +763,74 @@ class ScraplingMCPServer:
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by `google_search` takes priority over the referer set here if used together._ :param extra_headers: A dictionary of extra headers to add to the request. _The referer set by `google_search` takes priority over the referer set here if used together._
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. :param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
:param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings. :param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings.
:param session_id: Optional session ID from open_session. If provided, reuses the existing browser session instead of creating a new one.
""" """
async with AsyncStealthySession( if session_id:
wait=wait, entry = self._get_session(session_id, "stealthy")
proxy=proxy, fetch_kwargs = _build_fetch_kwargs(
locale=locale, wait=wait,
cdp_url=cdp_url, timeout=timeout,
timeout=timeout, google_search=google_search,
cookies=cookies, extra_headers=extra_headers,
headless=headless, disable_resources=disable_resources,
useragent=useragent, wait_selector=wait_selector,
timezone_id=timezone_id, wait_selector_state=wait_selector_state,
real_chrome=real_chrome, network_idle=network_idle,
hide_canvas=hide_canvas, proxy=proxy,
allow_webgl=allow_webgl, solve_cloudflare=solve_cloudflare,
network_idle=network_idle, )
block_webrtc=block_webrtc, tasks = [entry.session.fetch(url, **fetch_kwargs) for url in urls]
wait_selector=wait_selector,
google_search=google_search,
extra_headers=extra_headers,
additional_args=additional_args,
solve_cloudflare=solve_cloudflare,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
) as session:
tasks = [session.fetch(url) for url in urls]
responses = await gather(*tasks) responses = await gather(*tasks)
return [_translate_response(page, extraction_type, css_selector, main_content_only) for page in responses] else:
async with AsyncStealthySession(
wait=wait,
proxy=proxy,
locale=locale,
cdp_url=cdp_url,
timeout=timeout,
cookies=cookies,
headless=headless,
useragent=useragent,
timezone_id=timezone_id,
real_chrome=real_chrome,
hide_canvas=hide_canvas,
allow_webgl=allow_webgl,
network_idle=network_idle,
block_webrtc=block_webrtc,
wait_selector=wait_selector,
google_search=google_search,
extra_headers=extra_headers,
additional_args=additional_args,
solve_cloudflare=solve_cloudflare,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
) as session:
tasks = [session.fetch(url) for url in urls]
responses = await gather(*tasks)
return [_translate_response(page, extraction_type, css_selector, main_content_only) for page in responses]
def serve(self, http: bool, host: str, port: int): def serve(self, http: bool, host: str, port: int):
"""Serve the MCP server.""" """Serve the MCP server."""
server = FastMCP(name="Scrapling", host=host, port=port) server = FastMCP(name="Scrapling", host=host, port=port)
# Session management tools
server.add_tool(self.open_session, title="open_session", structured_output=True)
server.add_tool(self.close_session, title="close_session", structured_output=True)
server.add_tool(self.list_sessions, title="list_sessions", structured_output=True)
# HTTP tools
server.add_tool(self.get, title="get", description=self.get.__doc__, structured_output=True) server.add_tool(self.get, title="get", description=self.get.__doc__, structured_output=True)
server.add_tool(self.bulk_get, title="bulk_get", description=self.bulk_get.__doc__, structured_output=True) server.add_tool(self.bulk_get, title="bulk_get", description=self.bulk_get.__doc__, structured_output=True)
# Dynamic browser tools
server.add_tool(self.fetch, title="fetch", description=self.fetch.__doc__, structured_output=True) server.add_tool(self.fetch, title="fetch", description=self.fetch.__doc__, structured_output=True)
server.add_tool( server.add_tool(
self.bulk_fetch, title="bulk_fetch", description=self.bulk_fetch.__doc__, structured_output=True self.bulk_fetch, title="bulk_fetch", description=self.bulk_fetch.__doc__, structured_output=True
) )
# Stealthy browser tools
server.add_tool( server.add_tool(
self.stealthy_fetch, title="stealthy_fetch", description=self.stealthy_fetch.__doc__, structured_output=True self.stealthy_fetch,
title="stealthy_fetch",
description=self.stealthy_fetch.__doc__,
structured_output=True,
) )
server.add_tool( server.add_tool(
self.bulk_stealthy_fetch, self.bulk_stealthy_fetch,
+120 -1
View File
@@ -1,7 +1,14 @@
import pytest import pytest
import pytest_httpbin import pytest_httpbin
from scrapling.core.ai import ScraplingMCPServer, ResponseModel, _normalize_credentials from scrapling.core.ai import (
ScraplingMCPServer,
ResponseModel,
SessionInfo,
SessionCreatedModel,
SessionClosedModel,
_normalize_credentials,
)
@pytest_httpbin.use_class_based_httpbin @pytest_httpbin.use_class_based_httpbin
@@ -59,6 +66,118 @@ class TestMCPServer:
assert all(isinstance(r, ResponseModel) for r in result) assert all(isinstance(r, ResponseModel) for r in result)
@pytest_httpbin.use_class_based_httpbin
class TestSessionManagement:
"""Test persistent browser session management"""
@pytest.fixture(scope="class")
def test_url(self, httpbin):
return f"{httpbin.url}/html"
@pytest.fixture
def server(self):
return ScraplingMCPServer()
@pytest.mark.asyncio
async def test_open_and_close_session(self, server):
"""Test opening and closing a dynamic session"""
result = await server.open_session(session_type="dynamic", headless=True)
assert isinstance(result, SessionCreatedModel)
assert result.session_type == "dynamic"
assert result.is_alive is True
session_id = result.session_id
# Close the session
closed = await server.close_session(session_id)
assert isinstance(closed, SessionClosedModel)
assert closed.session_id == session_id
@pytest.mark.asyncio
async def test_list_sessions(self, server):
"""Test listing sessions"""
# Initially empty
sessions = await server.list_sessions()
assert sessions == []
# Open a session
result = await server.open_session(session_type="dynamic", headless=True)
session_id = result.session_id
# List should show it
sessions = await server.list_sessions()
assert len(sessions) == 1
assert isinstance(sessions[0], SessionInfo)
assert sessions[0].session_id == session_id
assert sessions[0].session_type == "dynamic"
assert sessions[0].is_alive is True
# Cleanup
await server.close_session(session_id)
@pytest.mark.asyncio
async def test_fetch_with_session(self, server, test_url):
"""Test fetching with a persistent dynamic session"""
result = await server.open_session(session_type="dynamic", headless=True)
session_id = result.session_id
# Fetch using the session
response = await server.fetch(url=test_url, session_id=session_id)
assert isinstance(response, ResponseModel)
assert response.status == 200
# Fetch again with the same session (reuse)
response2 = await server.fetch(url=test_url, session_id=session_id)
assert isinstance(response2, ResponseModel)
assert response2.status == 200
await server.close_session(session_id)
@pytest.mark.asyncio
async def test_bulk_fetch_with_session(self, server, test_url):
"""Test bulk fetching with a persistent dynamic session"""
result = await server.open_session(session_type="dynamic", headless=True, max_pages=5)
session_id = result.session_id
responses = await server.bulk_fetch(urls=[test_url, test_url], session_id=session_id)
assert len(responses) == 2
assert all(isinstance(r, ResponseModel) for r in responses)
await server.close_session(session_id)
@pytest.mark.asyncio
async def test_session_type_mismatch(self, server, test_url):
"""Test that using a dynamic session with stealthy_fetch raises an error"""
result = await server.open_session(session_type="dynamic", headless=True)
session_id = result.session_id
with pytest.raises(ValueError, match="'dynamic' session"):
await server.stealthy_fetch(url=test_url, session_id=session_id)
await server.close_session(session_id)
@pytest.mark.asyncio
async def test_close_nonexistent_session(self, server):
"""Test closing a session that doesn't exist"""
with pytest.raises(ValueError, match="not found"):
await server.close_session("nonexistent")
@pytest.mark.asyncio
async def test_fetch_with_nonexistent_session(self, server, test_url):
"""Test fetching with a session ID that doesn't exist"""
with pytest.raises(ValueError, match="not found"):
await server.fetch(url=test_url, session_id="nonexistent")
@pytest.mark.asyncio
async def test_fetch_with_closed_session(self, server, test_url):
"""Test fetching with a session that has been closed"""
result = await server.open_session(session_type="dynamic", headless=True)
session_id = result.session_id
await server.close_session(session_id)
with pytest.raises(ValueError, match="not found"):
await server.fetch(url=test_url, session_id=session_id)
class TestNormalizeCredentials: class TestNormalizeCredentials:
"""Test the _normalize_credentials helper""" """Test the _normalize_credentials helper"""