Merge branch 'dev' into main

This commit is contained in:
Karim shoair
2026-04-17 20:31:43 +02:00
committed by GitHub
10 changed files with 52 additions and 14 deletions
+2 -2
View File
@@ -1,7 +1,7 @@
--- ---
name: scrapling-official name: scrapling-official
description: Scrape web pages using Scrapling with anti-bot bypass (like Cloudflare Turnstile), stealth headless browsing, spiders framework, adaptive scraping, and JavaScript rendering. Use when asked to scrape, crawl, or extract data from websites; web_fetch fails; the site has anti-bot protections; write Python code to scrape/crawl; or write spiders. description: Scrape web pages using Scrapling with anti-bot bypass (like Cloudflare Turnstile), stealth headless browsing, spiders framework, adaptive scraping, and JavaScript rendering. Use when asked to scrape, crawl, or extract data from websites; web_fetch fails; the site has anti-bot protections; write Python code to scrape/crawl; or write spiders.
version: "0.4.6" version: "0.4.7"
license: Complete terms in LICENSE.txt license: Complete terms in LICENSE.txt
metadata: metadata:
homepage: "https://scrapling.readthedocs.io/en/latest/index.html" homepage: "https://scrapling.readthedocs.io/en/latest/index.html"
@@ -40,7 +40,7 @@ Blazing fast crawls with real-time stats and streaming. Built by Web Scrapers fo
Create a virtual Python environment through any way available, like `venv`, then inside the environment do: Create a virtual Python environment through any way available, like `venv`, then inside the environment do:
`pip install "scrapling[all]>=0.4.6"` `pip install "scrapling[all]>=0.4.7"`
Then do this to download all the browsers' dependencies: Then do this to download all the browsers' dependencies:
@@ -9,7 +9,7 @@ All examples collect **all 100 quotes across 10 pages**.
Make sure Scrapling is installed: Make sure Scrapling is installed:
```bash ```bash
pip install "scrapling[all]>=0.4.6" pip install "scrapling[all]>=0.4.7"
scrapling install --force scrapling install --force
``` ```
+3 -3
View File
@@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta"
[project] [project]
name = "scrapling" name = "scrapling"
# Static version instead of a dynamic version so we can get better layer caching while building docker, check the docker file to understand # Static version instead of a dynamic version so we can get better layer caching while building docker, check the docker file to understand
version = "0.4.6" version = "0.4.7"
description = "Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!" description = "Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!"
readme = {file = "README.md", content-type = "text/markdown"} readme = {file = "README.md", content-type = "text/markdown"}
license = {file = "LICENSE"} license = {file = "LICENSE"}
@@ -77,8 +77,8 @@ fetchers = [
"patchright==1.58.2", "patchright==1.58.2",
"browserforge>=1.2.4", "browserforge>=1.2.4",
"apify-fingerprint-datapoints>=0.12.0", "apify-fingerprint-datapoints>=0.12.0",
"msgspec>=0.21.0", "msgspec>=0.21.1",
"anyio>=4.12.1", "anyio>=4.13.0",
"protego>=0.6.0", "protego>=0.6.0",
] ]
ai = [ ai = [
+1 -1
View File
@@ -1,5 +1,5 @@
__author__ = "Karim Shoair (karim.shoair@pm.me)" __author__ = "Karim Shoair (karim.shoair@pm.me)"
__version__ = "0.4.6" __version__ = "0.4.7"
__copyright__ = "Copyright (c) 2024 Karim Shoair" __copyright__ = "Copyright (c) 2024 Karim Shoair"
from typing import Any, TYPE_CHECKING from typing import Any, TYPE_CHECKING
+8 -1
View File
@@ -123,6 +123,7 @@ class ScraplingMCPServer:
async def open_session( async def open_session(
self, self,
session_type: SessionType, session_type: SessionType,
session_id: Optional[str] = None,
headless: bool = True, headless: bool = True,
google_search: bool = True, google_search: bool = True,
real_chrome: bool = False, real_chrome: bool = False,
@@ -152,6 +153,7 @@ class ScraplingMCPServer:
Use close_session to close the session when done, and list_sessions to see all active sessions. Use close_session to close the session when done, and list_sessions to see all active sessions.
:param session_type: The type of session to open. Use "dynamic" for standard Playwright browser, or "stealthy" for anti-bot bypass with fingerprint spoofing. :param session_type: The type of session to open. Use "dynamic" for standard Playwright browser, or "stealthy" for anti-bot bypass with fingerprint spoofing.
:param session_id: Optional custom session ID. If not provided, a random 12-character hex ID will be generated. Useful for naming sessions for easier management.
:param headless: Run the browser in headless/hidden (default), or headful/visible mode. :param headless: Run the browser in headless/hidden (default), or headful/visible mode.
:param google_search: Enabled by default, Scrapling will set a Google referer header. :param google_search: Enabled by default, Scrapling will set a Google referer header.
:param real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it. :param real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it.
@@ -175,6 +177,12 @@ class ScraplingMCPServer:
:param solve_cloudflare: (Stealthy only) Solves all types of the Cloudflare's Turnstile/Interstitial challenges. :param solve_cloudflare: (Stealthy only) Solves all types of the Cloudflare's Turnstile/Interstitial challenges.
:param additional_args: (Stealthy only) Additional arguments to be passed to Playwright's context as additional settings. :param additional_args: (Stealthy only) Additional arguments to be passed to Playwright's context as additional settings.
""" """
session_id = session_id or uuid4().hex[:12]
if session_id in self._sessions:
raise ValueError(
f"Session '{session_id}' already exists. Use a different ID or close the existing session first."
)
common_kwargs: Dict[str, Any] = dict( common_kwargs: Dict[str, Any] = dict(
wait=wait, wait=wait,
proxy=proxy, proxy=proxy,
@@ -211,7 +219,6 @@ class ScraplingMCPServer:
await session.start() await session.start()
session_id = uuid4().hex[:12]
entry = _SessionEntry(session=session, session_type=session_type) entry = _SessionEntry(session=session, session_type=session_type)
self._sessions[session_id] = entry self._sessions[session_id] = entry
+12 -2
View File
@@ -718,8 +718,13 @@ class FetcherSession:
config["selector_config"] = self.selector_config config["selector_config"] = self.selector_config
config["proxy_rotator"] = self._proxy_rotator config["proxy_rotator"] = self._proxy_rotator
self._client = _SyncSessionLogic(**config) self._client = _SyncSessionLogic(**config)
try:
result = self._client.__enter__()
except Exception:
self._client = None
raise
self._is_alive = True self._is_alive = True
return self._client.__enter__() return result
raise RuntimeError("This FetcherSession instance already has an active synchronous session.") raise RuntimeError("This FetcherSession instance already has an active synchronous session.")
def __exit__(self, exc_type, exc_val, exc_tb): def __exit__(self, exc_type, exc_val, exc_tb):
@@ -739,8 +744,13 @@ class FetcherSession:
config["selector_config"] = self.selector_config config["selector_config"] = self.selector_config
config["proxy_rotator"] = self._proxy_rotator config["proxy_rotator"] = self._proxy_rotator
self._client = _ASyncSessionLogic(**config) self._client = _ASyncSessionLogic(**config)
try:
result = await self._client.__aenter__()
except Exception:
self._client = None
raise
self._is_alive = True self._is_alive = True
return await self._client.__aenter__() return result
raise RuntimeError("This FetcherSession instance already has an active asynchronous session.") raise RuntimeError("This FetcherSession instance already has an active asynchronous session.")
async def __aexit__(self, exc_type, exc_val, exc_tb): async def __aexit__(self, exc_type, exc_val, exc_tb):
+3 -1
View File
@@ -93,7 +93,9 @@ class SessionManager:
async def close(self) -> None: async def close(self) -> None:
"""Close all registered sessions.""" """Close all registered sessions."""
for session in self._sessions.values(): for sid, session in self._sessions.items():
if sid in self._lazy_sessions and not session._is_alive:
continue
_ = await session.__aexit__(None, None, None) _ = await session.__aexit__(None, None, None)
self._started = False self._started = False
+2 -2
View File
@@ -14,12 +14,12 @@
"mimeType": "image/png" "mimeType": "image/png"
} }
], ],
"version": "0.4.6", "version": "0.4.7",
"packages": [ "packages": [
{ {
"registryType": "pypi", "registryType": "pypi",
"identifier": "scrapling", "identifier": "scrapling",
"version": "0.4.6", "version": "0.4.7",
"runtimeHint": "uvx", "runtimeHint": "uvx",
"packageArguments": [ "packageArguments": [
{ {
+1 -1
View File
@@ -1,6 +1,6 @@
[metadata] [metadata]
name = scrapling name = scrapling
version = 0.4.6 version = 0.4.7
author = Karim Shoair author = Karim Shoair
author_email = karim.shoair@pm.me author_email = karim.shoair@pm.me
description = Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be! description = Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!
+19
View File
@@ -177,6 +177,25 @@ class TestSessionManagement:
with pytest.raises(ValueError, match="not found"): with pytest.raises(ValueError, match="not found"):
await server.fetch(url=test_url, session_id=session_id) await server.fetch(url=test_url, session_id=session_id)
@pytest.mark.asyncio
async def test_open_session_with_custom_id(self, server):
"""Test opening a session with a custom session_id"""
result = await server.open_session(session_type="dynamic", session_id="my-session", headless=True)
assert isinstance(result, SessionCreatedModel)
assert result.session_id == "my-session"
await server.close_session("my-session")
@pytest.mark.asyncio
async def test_open_session_duplicate_id_raises(self, server):
"""Test that opening a session with a duplicate session_id raises an error"""
await server.open_session(session_type="dynamic", session_id="dupe", headless=True)
with pytest.raises(ValueError, match="already exists"):
await server.open_session(session_type="dynamic", session_id="dupe", headless=True)
await server.close_session("dupe")
class TestNormalizeCredentials: class TestNormalizeCredentials:
"""Test the _normalize_credentials helper""" """Test the _normalize_credentials helper"""