Merge branch 'dev' into main
This commit is contained in:
@@ -1,7 +1,7 @@
|
||||
---
|
||||
name: scrapling-official
|
||||
description: Scrape web pages using Scrapling with anti-bot bypass (like Cloudflare Turnstile), stealth headless browsing, spiders framework, adaptive scraping, and JavaScript rendering. Use when asked to scrape, crawl, or extract data from websites; web_fetch fails; the site has anti-bot protections; write Python code to scrape/crawl; or write spiders.
|
||||
version: "0.4.6"
|
||||
version: "0.4.7"
|
||||
license: Complete terms in LICENSE.txt
|
||||
metadata:
|
||||
homepage: "https://scrapling.readthedocs.io/en/latest/index.html"
|
||||
@@ -40,7 +40,7 @@ Blazing fast crawls with real-time stats and streaming. Built by Web Scrapers fo
|
||||
|
||||
Create a virtual Python environment through any way available, like `venv`, then inside the environment do:
|
||||
|
||||
`pip install "scrapling[all]>=0.4.6"`
|
||||
`pip install "scrapling[all]>=0.4.7"`
|
||||
|
||||
Then do this to download all the browsers' dependencies:
|
||||
|
||||
|
||||
@@ -9,7 +9,7 @@ All examples collect **all 100 quotes across 10 pages**.
|
||||
Make sure Scrapling is installed:
|
||||
|
||||
```bash
|
||||
pip install "scrapling[all]>=0.4.6"
|
||||
pip install "scrapling[all]>=0.4.7"
|
||||
scrapling install --force
|
||||
```
|
||||
|
||||
|
||||
+3
-3
@@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta"
|
||||
[project]
|
||||
name = "scrapling"
|
||||
# Static version instead of a dynamic version so we can get better layer caching while building docker, check the docker file to understand
|
||||
version = "0.4.6"
|
||||
version = "0.4.7"
|
||||
description = "Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!"
|
||||
readme = {file = "README.md", content-type = "text/markdown"}
|
||||
license = {file = "LICENSE"}
|
||||
@@ -77,8 +77,8 @@ fetchers = [
|
||||
"patchright==1.58.2",
|
||||
"browserforge>=1.2.4",
|
||||
"apify-fingerprint-datapoints>=0.12.0",
|
||||
"msgspec>=0.21.0",
|
||||
"anyio>=4.12.1",
|
||||
"msgspec>=0.21.1",
|
||||
"anyio>=4.13.0",
|
||||
"protego>=0.6.0",
|
||||
]
|
||||
ai = [
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
__author__ = "Karim Shoair (karim.shoair@pm.me)"
|
||||
__version__ = "0.4.6"
|
||||
__version__ = "0.4.7"
|
||||
__copyright__ = "Copyright (c) 2024 Karim Shoair"
|
||||
|
||||
from typing import Any, TYPE_CHECKING
|
||||
|
||||
@@ -123,6 +123,7 @@ class ScraplingMCPServer:
|
||||
async def open_session(
|
||||
self,
|
||||
session_type: SessionType,
|
||||
session_id: Optional[str] = None,
|
||||
headless: bool = True,
|
||||
google_search: bool = True,
|
||||
real_chrome: bool = False,
|
||||
@@ -152,6 +153,7 @@ class ScraplingMCPServer:
|
||||
Use close_session to close the session when done, and list_sessions to see all active sessions.
|
||||
|
||||
:param session_type: The type of session to open. Use "dynamic" for standard Playwright browser, or "stealthy" for anti-bot bypass with fingerprint spoofing.
|
||||
:param session_id: Optional custom session ID. If not provided, a random 12-character hex ID will be generated. Useful for naming sessions for easier management.
|
||||
:param headless: Run the browser in headless/hidden (default), or headful/visible mode.
|
||||
:param google_search: Enabled by default, Scrapling will set a Google referer header.
|
||||
:param real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it.
|
||||
@@ -175,6 +177,12 @@ class ScraplingMCPServer:
|
||||
:param solve_cloudflare: (Stealthy only) Solves all types of the Cloudflare's Turnstile/Interstitial challenges.
|
||||
:param additional_args: (Stealthy only) Additional arguments to be passed to Playwright's context as additional settings.
|
||||
"""
|
||||
session_id = session_id or uuid4().hex[:12]
|
||||
if session_id in self._sessions:
|
||||
raise ValueError(
|
||||
f"Session '{session_id}' already exists. Use a different ID or close the existing session first."
|
||||
)
|
||||
|
||||
common_kwargs: Dict[str, Any] = dict(
|
||||
wait=wait,
|
||||
proxy=proxy,
|
||||
@@ -211,7 +219,6 @@ class ScraplingMCPServer:
|
||||
|
||||
await session.start()
|
||||
|
||||
session_id = uuid4().hex[:12]
|
||||
entry = _SessionEntry(session=session, session_type=session_type)
|
||||
self._sessions[session_id] = entry
|
||||
|
||||
|
||||
@@ -718,8 +718,13 @@ class FetcherSession:
|
||||
config["selector_config"] = self.selector_config
|
||||
config["proxy_rotator"] = self._proxy_rotator
|
||||
self._client = _SyncSessionLogic(**config)
|
||||
try:
|
||||
result = self._client.__enter__()
|
||||
except Exception:
|
||||
self._client = None
|
||||
raise
|
||||
self._is_alive = True
|
||||
return self._client.__enter__()
|
||||
return result
|
||||
raise RuntimeError("This FetcherSession instance already has an active synchronous session.")
|
||||
|
||||
def __exit__(self, exc_type, exc_val, exc_tb):
|
||||
@@ -739,8 +744,13 @@ class FetcherSession:
|
||||
config["selector_config"] = self.selector_config
|
||||
config["proxy_rotator"] = self._proxy_rotator
|
||||
self._client = _ASyncSessionLogic(**config)
|
||||
try:
|
||||
result = await self._client.__aenter__()
|
||||
except Exception:
|
||||
self._client = None
|
||||
raise
|
||||
self._is_alive = True
|
||||
return await self._client.__aenter__()
|
||||
return result
|
||||
raise RuntimeError("This FetcherSession instance already has an active asynchronous session.")
|
||||
|
||||
async def __aexit__(self, exc_type, exc_val, exc_tb):
|
||||
|
||||
@@ -93,7 +93,9 @@ class SessionManager:
|
||||
|
||||
async def close(self) -> None:
|
||||
"""Close all registered sessions."""
|
||||
for session in self._sessions.values():
|
||||
for sid, session in self._sessions.items():
|
||||
if sid in self._lazy_sessions and not session._is_alive:
|
||||
continue
|
||||
_ = await session.__aexit__(None, None, None)
|
||||
|
||||
self._started = False
|
||||
|
||||
+2
-2
@@ -14,12 +14,12 @@
|
||||
"mimeType": "image/png"
|
||||
}
|
||||
],
|
||||
"version": "0.4.6",
|
||||
"version": "0.4.7",
|
||||
"packages": [
|
||||
{
|
||||
"registryType": "pypi",
|
||||
"identifier": "scrapling",
|
||||
"version": "0.4.6",
|
||||
"version": "0.4.7",
|
||||
"runtimeHint": "uvx",
|
||||
"packageArguments": [
|
||||
{
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[metadata]
|
||||
name = scrapling
|
||||
version = 0.4.6
|
||||
version = 0.4.7
|
||||
author = Karim Shoair
|
||||
author_email = karim.shoair@pm.me
|
||||
description = Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!
|
||||
|
||||
@@ -177,6 +177,25 @@ class TestSessionManagement:
|
||||
with pytest.raises(ValueError, match="not found"):
|
||||
await server.fetch(url=test_url, session_id=session_id)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_open_session_with_custom_id(self, server):
|
||||
"""Test opening a session with a custom session_id"""
|
||||
result = await server.open_session(session_type="dynamic", session_id="my-session", headless=True)
|
||||
assert isinstance(result, SessionCreatedModel)
|
||||
assert result.session_id == "my-session"
|
||||
|
||||
await server.close_session("my-session")
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_open_session_duplicate_id_raises(self, server):
|
||||
"""Test that opening a session with a duplicate session_id raises an error"""
|
||||
await server.open_session(session_type="dynamic", session_id="dupe", headless=True)
|
||||
|
||||
with pytest.raises(ValueError, match="already exists"):
|
||||
await server.open_session(session_type="dynamic", session_id="dupe", headless=True)
|
||||
|
||||
await server.close_session("dupe")
|
||||
|
||||
|
||||
class TestNormalizeCredentials:
|
||||
"""Test the _normalize_credentials helper"""
|
||||
|
||||
Reference in New Issue
Block a user