Merge branch 'dev' into main
This commit is contained in:
@@ -1,7 +1,7 @@
|
|||||||
---
|
---
|
||||||
name: scrapling-official
|
name: scrapling-official
|
||||||
description: Scrape web pages using Scrapling with anti-bot bypass (like Cloudflare Turnstile), stealth headless browsing, spiders framework, adaptive scraping, and JavaScript rendering. Use when asked to scrape, crawl, or extract data from websites; web_fetch fails; the site has anti-bot protections; write Python code to scrape/crawl; or write spiders.
|
description: Scrape web pages using Scrapling with anti-bot bypass (like Cloudflare Turnstile), stealth headless browsing, spiders framework, adaptive scraping, and JavaScript rendering. Use when asked to scrape, crawl, or extract data from websites; web_fetch fails; the site has anti-bot protections; write Python code to scrape/crawl; or write spiders.
|
||||||
version: "0.4.6"
|
version: "0.4.7"
|
||||||
license: Complete terms in LICENSE.txt
|
license: Complete terms in LICENSE.txt
|
||||||
metadata:
|
metadata:
|
||||||
homepage: "https://scrapling.readthedocs.io/en/latest/index.html"
|
homepage: "https://scrapling.readthedocs.io/en/latest/index.html"
|
||||||
@@ -40,7 +40,7 @@ Blazing fast crawls with real-time stats and streaming. Built by Web Scrapers fo
|
|||||||
|
|
||||||
Create a virtual Python environment through any way available, like `venv`, then inside the environment do:
|
Create a virtual Python environment through any way available, like `venv`, then inside the environment do:
|
||||||
|
|
||||||
`pip install "scrapling[all]>=0.4.6"`
|
`pip install "scrapling[all]>=0.4.7"`
|
||||||
|
|
||||||
Then do this to download all the browsers' dependencies:
|
Then do this to download all the browsers' dependencies:
|
||||||
|
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ All examples collect **all 100 quotes across 10 pages**.
|
|||||||
Make sure Scrapling is installed:
|
Make sure Scrapling is installed:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
pip install "scrapling[all]>=0.4.6"
|
pip install "scrapling[all]>=0.4.7"
|
||||||
scrapling install --force
|
scrapling install --force
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|||||||
+3
-3
@@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta"
|
|||||||
[project]
|
[project]
|
||||||
name = "scrapling"
|
name = "scrapling"
|
||||||
# Static version instead of a dynamic version so we can get better layer caching while building docker, check the docker file to understand
|
# Static version instead of a dynamic version so we can get better layer caching while building docker, check the docker file to understand
|
||||||
version = "0.4.6"
|
version = "0.4.7"
|
||||||
description = "Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!"
|
description = "Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!"
|
||||||
readme = {file = "README.md", content-type = "text/markdown"}
|
readme = {file = "README.md", content-type = "text/markdown"}
|
||||||
license = {file = "LICENSE"}
|
license = {file = "LICENSE"}
|
||||||
@@ -77,8 +77,8 @@ fetchers = [
|
|||||||
"patchright==1.58.2",
|
"patchright==1.58.2",
|
||||||
"browserforge>=1.2.4",
|
"browserforge>=1.2.4",
|
||||||
"apify-fingerprint-datapoints>=0.12.0",
|
"apify-fingerprint-datapoints>=0.12.0",
|
||||||
"msgspec>=0.21.0",
|
"msgspec>=0.21.1",
|
||||||
"anyio>=4.12.1",
|
"anyio>=4.13.0",
|
||||||
"protego>=0.6.0",
|
"protego>=0.6.0",
|
||||||
]
|
]
|
||||||
ai = [
|
ai = [
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
__author__ = "Karim Shoair (karim.shoair@pm.me)"
|
__author__ = "Karim Shoair (karim.shoair@pm.me)"
|
||||||
__version__ = "0.4.6"
|
__version__ = "0.4.7"
|
||||||
__copyright__ = "Copyright (c) 2024 Karim Shoair"
|
__copyright__ = "Copyright (c) 2024 Karim Shoair"
|
||||||
|
|
||||||
from typing import Any, TYPE_CHECKING
|
from typing import Any, TYPE_CHECKING
|
||||||
|
|||||||
@@ -123,6 +123,7 @@ class ScraplingMCPServer:
|
|||||||
async def open_session(
|
async def open_session(
|
||||||
self,
|
self,
|
||||||
session_type: SessionType,
|
session_type: SessionType,
|
||||||
|
session_id: Optional[str] = None,
|
||||||
headless: bool = True,
|
headless: bool = True,
|
||||||
google_search: bool = True,
|
google_search: bool = True,
|
||||||
real_chrome: bool = False,
|
real_chrome: bool = False,
|
||||||
@@ -152,6 +153,7 @@ class ScraplingMCPServer:
|
|||||||
Use close_session to close the session when done, and list_sessions to see all active sessions.
|
Use close_session to close the session when done, and list_sessions to see all active sessions.
|
||||||
|
|
||||||
:param session_type: The type of session to open. Use "dynamic" for standard Playwright browser, or "stealthy" for anti-bot bypass with fingerprint spoofing.
|
:param session_type: The type of session to open. Use "dynamic" for standard Playwright browser, or "stealthy" for anti-bot bypass with fingerprint spoofing.
|
||||||
|
:param session_id: Optional custom session ID. If not provided, a random 12-character hex ID will be generated. Useful for naming sessions for easier management.
|
||||||
:param headless: Run the browser in headless/hidden (default), or headful/visible mode.
|
:param headless: Run the browser in headless/hidden (default), or headful/visible mode.
|
||||||
:param google_search: Enabled by default, Scrapling will set a Google referer header.
|
:param google_search: Enabled by default, Scrapling will set a Google referer header.
|
||||||
:param real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it.
|
:param real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it.
|
||||||
@@ -175,6 +177,12 @@ class ScraplingMCPServer:
|
|||||||
:param solve_cloudflare: (Stealthy only) Solves all types of the Cloudflare's Turnstile/Interstitial challenges.
|
:param solve_cloudflare: (Stealthy only) Solves all types of the Cloudflare's Turnstile/Interstitial challenges.
|
||||||
:param additional_args: (Stealthy only) Additional arguments to be passed to Playwright's context as additional settings.
|
:param additional_args: (Stealthy only) Additional arguments to be passed to Playwright's context as additional settings.
|
||||||
"""
|
"""
|
||||||
|
session_id = session_id or uuid4().hex[:12]
|
||||||
|
if session_id in self._sessions:
|
||||||
|
raise ValueError(
|
||||||
|
f"Session '{session_id}' already exists. Use a different ID or close the existing session first."
|
||||||
|
)
|
||||||
|
|
||||||
common_kwargs: Dict[str, Any] = dict(
|
common_kwargs: Dict[str, Any] = dict(
|
||||||
wait=wait,
|
wait=wait,
|
||||||
proxy=proxy,
|
proxy=proxy,
|
||||||
@@ -211,7 +219,6 @@ class ScraplingMCPServer:
|
|||||||
|
|
||||||
await session.start()
|
await session.start()
|
||||||
|
|
||||||
session_id = uuid4().hex[:12]
|
|
||||||
entry = _SessionEntry(session=session, session_type=session_type)
|
entry = _SessionEntry(session=session, session_type=session_type)
|
||||||
self._sessions[session_id] = entry
|
self._sessions[session_id] = entry
|
||||||
|
|
||||||
|
|||||||
@@ -718,8 +718,13 @@ class FetcherSession:
|
|||||||
config["selector_config"] = self.selector_config
|
config["selector_config"] = self.selector_config
|
||||||
config["proxy_rotator"] = self._proxy_rotator
|
config["proxy_rotator"] = self._proxy_rotator
|
||||||
self._client = _SyncSessionLogic(**config)
|
self._client = _SyncSessionLogic(**config)
|
||||||
|
try:
|
||||||
|
result = self._client.__enter__()
|
||||||
|
except Exception:
|
||||||
|
self._client = None
|
||||||
|
raise
|
||||||
self._is_alive = True
|
self._is_alive = True
|
||||||
return self._client.__enter__()
|
return result
|
||||||
raise RuntimeError("This FetcherSession instance already has an active synchronous session.")
|
raise RuntimeError("This FetcherSession instance already has an active synchronous session.")
|
||||||
|
|
||||||
def __exit__(self, exc_type, exc_val, exc_tb):
|
def __exit__(self, exc_type, exc_val, exc_tb):
|
||||||
@@ -739,8 +744,13 @@ class FetcherSession:
|
|||||||
config["selector_config"] = self.selector_config
|
config["selector_config"] = self.selector_config
|
||||||
config["proxy_rotator"] = self._proxy_rotator
|
config["proxy_rotator"] = self._proxy_rotator
|
||||||
self._client = _ASyncSessionLogic(**config)
|
self._client = _ASyncSessionLogic(**config)
|
||||||
|
try:
|
||||||
|
result = await self._client.__aenter__()
|
||||||
|
except Exception:
|
||||||
|
self._client = None
|
||||||
|
raise
|
||||||
self._is_alive = True
|
self._is_alive = True
|
||||||
return await self._client.__aenter__()
|
return result
|
||||||
raise RuntimeError("This FetcherSession instance already has an active asynchronous session.")
|
raise RuntimeError("This FetcherSession instance already has an active asynchronous session.")
|
||||||
|
|
||||||
async def __aexit__(self, exc_type, exc_val, exc_tb):
|
async def __aexit__(self, exc_type, exc_val, exc_tb):
|
||||||
|
|||||||
@@ -93,7 +93,9 @@ class SessionManager:
|
|||||||
|
|
||||||
async def close(self) -> None:
|
async def close(self) -> None:
|
||||||
"""Close all registered sessions."""
|
"""Close all registered sessions."""
|
||||||
for session in self._sessions.values():
|
for sid, session in self._sessions.items():
|
||||||
|
if sid in self._lazy_sessions and not session._is_alive:
|
||||||
|
continue
|
||||||
_ = await session.__aexit__(None, None, None)
|
_ = await session.__aexit__(None, None, None)
|
||||||
|
|
||||||
self._started = False
|
self._started = False
|
||||||
|
|||||||
+2
-2
@@ -14,12 +14,12 @@
|
|||||||
"mimeType": "image/png"
|
"mimeType": "image/png"
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
"version": "0.4.6",
|
"version": "0.4.7",
|
||||||
"packages": [
|
"packages": [
|
||||||
{
|
{
|
||||||
"registryType": "pypi",
|
"registryType": "pypi",
|
||||||
"identifier": "scrapling",
|
"identifier": "scrapling",
|
||||||
"version": "0.4.6",
|
"version": "0.4.7",
|
||||||
"runtimeHint": "uvx",
|
"runtimeHint": "uvx",
|
||||||
"packageArguments": [
|
"packageArguments": [
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
[metadata]
|
[metadata]
|
||||||
name = scrapling
|
name = scrapling
|
||||||
version = 0.4.6
|
version = 0.4.7
|
||||||
author = Karim Shoair
|
author = Karim Shoair
|
||||||
author_email = karim.shoair@pm.me
|
author_email = karim.shoair@pm.me
|
||||||
description = Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!
|
description = Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!
|
||||||
|
|||||||
@@ -177,6 +177,25 @@ class TestSessionManagement:
|
|||||||
with pytest.raises(ValueError, match="not found"):
|
with pytest.raises(ValueError, match="not found"):
|
||||||
await server.fetch(url=test_url, session_id=session_id)
|
await server.fetch(url=test_url, session_id=session_id)
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_open_session_with_custom_id(self, server):
|
||||||
|
"""Test opening a session with a custom session_id"""
|
||||||
|
result = await server.open_session(session_type="dynamic", session_id="my-session", headless=True)
|
||||||
|
assert isinstance(result, SessionCreatedModel)
|
||||||
|
assert result.session_id == "my-session"
|
||||||
|
|
||||||
|
await server.close_session("my-session")
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_open_session_duplicate_id_raises(self, server):
|
||||||
|
"""Test that opening a session with a duplicate session_id raises an error"""
|
||||||
|
await server.open_session(session_type="dynamic", session_id="dupe", headless=True)
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="already exists"):
|
||||||
|
await server.open_session(session_type="dynamic", session_id="dupe", headless=True)
|
||||||
|
|
||||||
|
await server.close_session("dupe")
|
||||||
|
|
||||||
|
|
||||||
class TestNormalizeCredentials:
|
class TestNormalizeCredentials:
|
||||||
"""Test the _normalize_credentials helper"""
|
"""Test the _normalize_credentials helper"""
|
||||||
|
|||||||
Reference in New Issue
Block a user