feat(browsers): add a new feature to block ads
This is working by aborting all requests to known ads domains.
This commit is contained in:
@@ -309,6 +309,11 @@ def _common_browser_options(f):
|
|||||||
default=True,
|
default=True,
|
||||||
help="Run browser in headless mode (default: True)",
|
help="Run browser in headless mode (default: True)",
|
||||||
),
|
),
|
||||||
|
option(
|
||||||
|
"--block-ads/--no-block-ads",
|
||||||
|
default=False,
|
||||||
|
help="Block requests to known ad and tracker domains (default: False)",
|
||||||
|
),
|
||||||
]
|
]
|
||||||
for decorator in decorators:
|
for decorator in decorators:
|
||||||
f = decorator(f)
|
f = decorator(f)
|
||||||
@@ -498,6 +503,7 @@ def __build_browser_kwargs(
|
|||||||
real_chrome,
|
real_chrome,
|
||||||
proxy,
|
proxy,
|
||||||
parsed_headers,
|
parsed_headers,
|
||||||
|
block_ads,
|
||||||
) -> Dict[str, Any]:
|
) -> Dict[str, Any]:
|
||||||
"""Build shared kwargs dict for browser-based commands."""
|
"""Build shared kwargs dict for browser-based commands."""
|
||||||
kwargs: Dict[str, Any] = {
|
kwargs: Dict[str, Any] = {
|
||||||
@@ -507,6 +513,7 @@ def __build_browser_kwargs(
|
|||||||
"timeout": timeout,
|
"timeout": timeout,
|
||||||
"locale": locale,
|
"locale": locale,
|
||||||
"real_chrome": real_chrome,
|
"real_chrome": real_chrome,
|
||||||
|
"block_ads": block_ads,
|
||||||
}
|
}
|
||||||
if wait > 0:
|
if wait > 0:
|
||||||
kwargs["wait"] = wait
|
kwargs["wait"] = wait
|
||||||
@@ -538,6 +545,7 @@ def fetch(
|
|||||||
proxy,
|
proxy,
|
||||||
extra_headers,
|
extra_headers,
|
||||||
ai_targeted,
|
ai_targeted,
|
||||||
|
block_ads,
|
||||||
):
|
):
|
||||||
"""Opens up a browser and fetch content using DynamicFetcher."""
|
"""Opens up a browser and fetch content using DynamicFetcher."""
|
||||||
parsed_headers, _ = _ParseHeaders(extra_headers, False)
|
parsed_headers, _ = _ParseHeaders(extra_headers, False)
|
||||||
@@ -552,6 +560,7 @@ def fetch(
|
|||||||
real_chrome,
|
real_chrome,
|
||||||
proxy,
|
proxy,
|
||||||
parsed_headers,
|
parsed_headers,
|
||||||
|
block_ads,
|
||||||
)
|
)
|
||||||
from scrapling.fetchers import DynamicFetcher
|
from scrapling.fetchers import DynamicFetcher
|
||||||
|
|
||||||
@@ -597,6 +606,7 @@ def stealthy_fetch(
|
|||||||
allow_webgl,
|
allow_webgl,
|
||||||
hide_canvas,
|
hide_canvas,
|
||||||
ai_targeted,
|
ai_targeted,
|
||||||
|
block_ads,
|
||||||
):
|
):
|
||||||
"""Opens up a browser with advanced stealth features and fetch content using StealthyFetcher."""
|
"""Opens up a browser with advanced stealth features and fetch content using StealthyFetcher."""
|
||||||
parsed_headers, _ = _ParseHeaders(extra_headers, False)
|
parsed_headers, _ = _ParseHeaders(extra_headers, False)
|
||||||
@@ -611,6 +621,7 @@ def stealthy_fetch(
|
|||||||
real_chrome,
|
real_chrome,
|
||||||
proxy,
|
proxy,
|
||||||
parsed_headers,
|
parsed_headers,
|
||||||
|
block_ads,
|
||||||
)
|
)
|
||||||
kwargs.update(
|
kwargs.update(
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -58,6 +58,7 @@ _FETCH_PARAMS = {
|
|||||||
"cdp_url": Optional[str],
|
"cdp_url": Optional[str],
|
||||||
"useragent": Optional[str],
|
"useragent": Optional[str],
|
||||||
"extra_flags": Optional[List[str]],
|
"extra_flags": Optional[List[str]],
|
||||||
|
"block_ads": bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
_STEALTHY_FETCH_PARAMS = {
|
_STEALTHY_FETCH_PARAMS = {
|
||||||
@@ -84,6 +85,7 @@ _STEALTHY_FETCH_PARAMS = {
|
|||||||
"cdp_url": Optional[str],
|
"cdp_url": Optional[str],
|
||||||
"useragent": Optional[str],
|
"useragent": Optional[str],
|
||||||
"extra_flags": Optional[List[str]],
|
"extra_flags": Optional[List[str]],
|
||||||
|
"block_ads": bool,
|
||||||
"allow_webgl": bool,
|
"allow_webgl": bool,
|
||||||
"hide_canvas": bool,
|
"hide_canvas": bool,
|
||||||
"block_webrtc": bool,
|
"block_webrtc": bool,
|
||||||
|
|||||||
@@ -88,6 +88,7 @@ class PlaywrightSession(TypedDict, total=False):
|
|||||||
useragent: Optional[str]
|
useragent: Optional[str]
|
||||||
extra_flags: Optional[List[str]]
|
extra_flags: Optional[List[str]]
|
||||||
blocked_domains: Optional[Set[str]]
|
blocked_domains: Optional[Set[str]]
|
||||||
|
block_ads: bool
|
||||||
retries: int
|
retries: int
|
||||||
retry_delay: int | float
|
retry_delay: int | float
|
||||||
capture_xhr: str | None
|
capture_xhr: str | None
|
||||||
|
|||||||
@@ -85,6 +85,7 @@ class PlaywrightConfig(Struct, kw_only=True, frozen=False, weakref=True):
|
|||||||
useragent: Optional[str] = None
|
useragent: Optional[str] = None
|
||||||
extra_flags: Optional[List[str]] = None
|
extra_flags: Optional[List[str]] = None
|
||||||
blocked_domains: Optional[Set[str]] = None
|
blocked_domains: Optional[Set[str]] = None
|
||||||
|
block_ads: bool = False
|
||||||
retries: RetriesCount = 3
|
retries: RetriesCount = 3
|
||||||
retry_delay: Seconds = 1
|
retry_delay: Seconds = 1
|
||||||
capture_xhr: str | None = None
|
capture_xhr: str | None = None
|
||||||
@@ -127,6 +128,14 @@ class PlaywrightConfig(Struct, kw_only=True, frozen=False, weakref=True):
|
|||||||
if validation_msg:
|
if validation_msg:
|
||||||
raise ValueError(validation_msg)
|
raise ValueError(validation_msg)
|
||||||
|
|
||||||
|
if self.block_ads:
|
||||||
|
from scrapling.engines.toolbelt.ad_domains import AD_DOMAINS
|
||||||
|
|
||||||
|
if self.blocked_domains:
|
||||||
|
self.blocked_domains = self.blocked_domains | set(AD_DOMAINS)
|
||||||
|
else:
|
||||||
|
self.blocked_domains = set(AD_DOMAINS)
|
||||||
|
|
||||||
|
|
||||||
class StealthConfig(PlaywrightConfig, kw_only=True, frozen=False, weakref=True):
|
class StealthConfig(PlaywrightConfig, kw_only=True, frozen=False, weakref=True):
|
||||||
allow_webgl: bool = True
|
allow_webgl: bool = True
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -19,6 +19,27 @@ class ProxyDict(Struct):
|
|||||||
password: str = ""
|
password: str = ""
|
||||||
|
|
||||||
|
|
||||||
|
def _is_domain_blocked(hostname: str, domains: frozenset) -> bool:
|
||||||
|
"""Check if a hostname matches any blocked domain using O(1) frozenset lookups.
|
||||||
|
|
||||||
|
Walks up the hostname's suffix chain: for "tracker.ads.doubleclick.net",
|
||||||
|
checks "tracker.ads.doubleclick.net", "ads.doubleclick.net", "doubleclick.net".
|
||||||
|
|
||||||
|
:param hostname: The hostname to check.
|
||||||
|
:param domains: A frozenset of blocked domain names.
|
||||||
|
:return: True if the hostname or any of its parent domains is in the blocked set.
|
||||||
|
"""
|
||||||
|
if hostname in domains:
|
||||||
|
return True
|
||||||
|
idx = hostname.find(".")
|
||||||
|
while idx != -1:
|
||||||
|
suffix = hostname[idx + 1 :]
|
||||||
|
if "." in suffix and suffix in domains:
|
||||||
|
return True
|
||||||
|
idx = hostname.find(".", idx + 1)
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
def create_intercept_handler(disable_resources: bool, blocked_domains: Optional[Set[str]] = None) -> Callable:
|
def create_intercept_handler(disable_resources: bool, blocked_domains: Optional[Set[str]] = None) -> Callable:
|
||||||
"""Create a route handler that blocks both resource types and specific domains.
|
"""Create a route handler that blocks both resource types and specific domains.
|
||||||
|
|
||||||
@@ -27,7 +48,7 @@ def create_intercept_handler(disable_resources: bool, blocked_domains: Optional[
|
|||||||
:return: A sync route handler function.
|
:return: A sync route handler function.
|
||||||
"""
|
"""
|
||||||
disabled_resources = EXTRA_RESOURCES if disable_resources else set()
|
disabled_resources = EXTRA_RESOURCES if disable_resources else set()
|
||||||
domains = blocked_domains or set()
|
domains = frozenset(blocked_domains) if blocked_domains else frozenset()
|
||||||
|
|
||||||
def handler(route: Route):
|
def handler(route: Route):
|
||||||
if route.request.resource_type in disabled_resources:
|
if route.request.resource_type in disabled_resources:
|
||||||
@@ -35,7 +56,7 @@ def create_intercept_handler(disable_resources: bool, blocked_domains: Optional[
|
|||||||
route.abort()
|
route.abort()
|
||||||
elif domains:
|
elif domains:
|
||||||
hostname = urlparse(route.request.url).hostname or ""
|
hostname = urlparse(route.request.url).hostname or ""
|
||||||
if any(hostname == d or hostname.endswith("." + d) for d in domains):
|
if _is_domain_blocked(hostname, domains):
|
||||||
log.debug(f'Blocking request to blocked domain "{hostname}" ({route.request.url})')
|
log.debug(f'Blocking request to blocked domain "{hostname}" ({route.request.url})')
|
||||||
route.abort()
|
route.abort()
|
||||||
else:
|
else:
|
||||||
@@ -54,7 +75,7 @@ def create_async_intercept_handler(disable_resources: bool, blocked_domains: Opt
|
|||||||
:return: An async route handler function.
|
:return: An async route handler function.
|
||||||
"""
|
"""
|
||||||
disabled_resources = EXTRA_RESOURCES if disable_resources else set()
|
disabled_resources = EXTRA_RESOURCES if disable_resources else set()
|
||||||
domains = blocked_domains or set()
|
domains = frozenset(blocked_domains) if blocked_domains else frozenset()
|
||||||
|
|
||||||
async def handler(route: async_Route):
|
async def handler(route: async_Route):
|
||||||
if route.request.resource_type in disabled_resources:
|
if route.request.resource_type in disabled_resources:
|
||||||
@@ -62,7 +83,7 @@ def create_async_intercept_handler(disable_resources: bool, blocked_domains: Opt
|
|||||||
await route.abort()
|
await route.abort()
|
||||||
elif domains:
|
elif domains:
|
||||||
hostname = urlparse(route.request.url).hostname or ""
|
hostname = urlparse(route.request.url).hostname or ""
|
||||||
if any(hostname == d or hostname.endswith("." + d) for d in domains):
|
if _is_domain_blocked(hostname, domains):
|
||||||
log.debug(f'Blocking request to blocked domain "{hostname}" ({route.request.url})')
|
log.debug(f'Blocking request to blocked domain "{hostname}" ({route.request.url})')
|
||||||
await route.abort()
|
await route.abort()
|
||||||
else:
|
else:
|
||||||
|
|||||||
@@ -15,6 +15,7 @@ class DynamicFetcher(BaseFetcher):
|
|||||||
:param headless: Run the browser in headless/hidden (default), or headful/visible mode.
|
:param headless: Run the browser in headless/hidden (default), or headful/visible mode.
|
||||||
:param disable_resources: Drop requests for unnecessary resources for a speed boost.
|
:param disable_resources: Drop requests for unnecessary resources for a speed boost.
|
||||||
:param blocked_domains: A set of domain names to block requests to. Subdomains are also matched (e.g., ``"example.com"`` blocks ``"sub.example.com"`` too).
|
:param blocked_domains: A set of domain names to block requests to. Subdomains are also matched (e.g., ``"example.com"`` blocks ``"sub.example.com"`` too).
|
||||||
|
:param block_ads: Block requests to ~3,500 known ad/tracking domains. Can be combined with ``blocked_domains``.
|
||||||
:param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it.
|
:param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it.
|
||||||
:param cookies: Set cookies for the next request.
|
:param cookies: Set cookies for the next request.
|
||||||
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
||||||
@@ -55,6 +56,7 @@ class DynamicFetcher(BaseFetcher):
|
|||||||
:param headless: Run the browser in headless/hidden (default), or headful/visible mode.
|
:param headless: Run the browser in headless/hidden (default), or headful/visible mode.
|
||||||
:param disable_resources: Drop requests for unnecessary resources for a speed boost.
|
:param disable_resources: Drop requests for unnecessary resources for a speed boost.
|
||||||
:param blocked_domains: A set of domain names to block requests to. Subdomains are also matched (e.g., ``"example.com"`` blocks ``"sub.example.com"`` too).
|
:param blocked_domains: A set of domain names to block requests to. Subdomains are also matched (e.g., ``"example.com"`` blocks ``"sub.example.com"`` too).
|
||||||
|
:param block_ads: Block requests to ~3,500 known ad/tracking domains. Can be combined with ``blocked_domains``.
|
||||||
:param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it.
|
:param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it.
|
||||||
:param cookies: Set cookies for the next request.
|
:param cookies: Set cookies for the next request.
|
||||||
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ class StealthyFetcher(BaseFetcher):
|
|||||||
:param disable_resources: Drop requests for unnecessary resources for a speed boost.
|
:param disable_resources: Drop requests for unnecessary resources for a speed boost.
|
||||||
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
|
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
|
||||||
:param blocked_domains: A set of domain names to block requests to. Subdomains are also matched (e.g., ``"example.com"`` blocks ``"sub.example.com"`` too).
|
:param blocked_domains: A set of domain names to block requests to. Subdomains are also matched (e.g., ``"example.com"`` blocks ``"sub.example.com"`` too).
|
||||||
|
:param block_ads: Block requests to ~3,500 known ad/tracking domains. Can be combined with ``blocked_domains``.
|
||||||
:param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it.
|
:param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it.
|
||||||
:param cookies: Set cookies for the next request.
|
:param cookies: Set cookies for the next request.
|
||||||
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
||||||
@@ -69,6 +70,7 @@ class StealthyFetcher(BaseFetcher):
|
|||||||
:param disable_resources: Drop requests for unnecessary resources for a speed boost.
|
:param disable_resources: Drop requests for unnecessary resources for a speed boost.
|
||||||
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
|
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
|
||||||
:param blocked_domains: A set of domain names to block requests to. Subdomains are also matched (e.g., ``"example.com"`` blocks ``"sub.example.com"`` too).
|
:param blocked_domains: A set of domain names to block requests to. Subdomains are also matched (e.g., ``"example.com"`` blocks ``"sub.example.com"`` too).
|
||||||
|
:param block_ads: Block requests to ~3,500 known ad/tracking domains. Can be combined with ``blocked_domains``.
|
||||||
:param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it.
|
:param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it.
|
||||||
:param cookies: Set cookies for the next request.
|
:param cookies: Set cookies for the next request.
|
||||||
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
||||||
|
|||||||
@@ -5,11 +5,9 @@ from scrapling.engines.toolbelt.navigation import (
|
|||||||
construct_proxy_dict,
|
construct_proxy_dict,
|
||||||
create_intercept_handler,
|
create_intercept_handler,
|
||||||
create_async_intercept_handler,
|
create_async_intercept_handler,
|
||||||
|
_is_domain_blocked,
|
||||||
)
|
)
|
||||||
from scrapling.engines.toolbelt.fingerprints import (
|
from scrapling.engines.toolbelt.fingerprints import get_os_name, generate_headers
|
||||||
get_os_name,
|
|
||||||
generate_headers
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture
|
@pytest.fixture
|
||||||
@@ -148,31 +146,19 @@ class TestConstructProxyDict:
|
|||||||
"""Test a basic proxy string"""
|
"""Test a basic proxy string"""
|
||||||
result = construct_proxy_dict("http://proxy.example.com:8080")
|
result = construct_proxy_dict("http://proxy.example.com:8080")
|
||||||
|
|
||||||
expected = {
|
expected = {"server": "http://proxy.example.com:8080", "username": "", "password": ""}
|
||||||
"server": "http://proxy.example.com:8080",
|
|
||||||
"username": "",
|
|
||||||
"password": ""
|
|
||||||
}
|
|
||||||
assert result == expected
|
assert result == expected
|
||||||
|
|
||||||
def test_proxy_string_with_auth(self):
|
def test_proxy_string_with_auth(self):
|
||||||
"""Test proxy string with authentication"""
|
"""Test proxy string with authentication"""
|
||||||
result = construct_proxy_dict("http://user:pass@proxy.example.com:8080")
|
result = construct_proxy_dict("http://user:pass@proxy.example.com:8080")
|
||||||
|
|
||||||
expected = {
|
expected = {"server": "http://proxy.example.com:8080", "username": "user", "password": "pass"}
|
||||||
"server": "http://proxy.example.com:8080",
|
|
||||||
"username": "user",
|
|
||||||
"password": "pass"
|
|
||||||
}
|
|
||||||
assert result == expected
|
assert result == expected
|
||||||
|
|
||||||
def test_proxy_dict_input(self):
|
def test_proxy_dict_input(self):
|
||||||
"""Test proxy dictionary input"""
|
"""Test proxy dictionary input"""
|
||||||
input_dict = {
|
input_dict = {"server": "http://proxy.example.com:8080", "username": "user", "password": "pass"}
|
||||||
"server": "http://proxy.example.com:8080",
|
|
||||||
"username": "user",
|
|
||||||
"password": "pass"
|
|
||||||
}
|
|
||||||
result = construct_proxy_dict(input_dict)
|
result = construct_proxy_dict(input_dict)
|
||||||
|
|
||||||
assert result == input_dict
|
assert result == input_dict
|
||||||
@@ -182,11 +168,7 @@ class TestConstructProxyDict:
|
|||||||
input_dict = {"server": "http://proxy.example.com:8080"}
|
input_dict = {"server": "http://proxy.example.com:8080"}
|
||||||
result = construct_proxy_dict(input_dict)
|
result = construct_proxy_dict(input_dict)
|
||||||
|
|
||||||
expected = {
|
expected = {"server": "http://proxy.example.com:8080", "username": "", "password": ""}
|
||||||
"server": "http://proxy.example.com:8080",
|
|
||||||
"username": "",
|
|
||||||
"password": ""
|
|
||||||
}
|
|
||||||
assert result == expected
|
assert result == expected
|
||||||
|
|
||||||
def test_invalid_proxy_string(self):
|
def test_invalid_proxy_string(self):
|
||||||
@@ -240,7 +222,7 @@ class TestResponse:
|
|||||||
cookies={"session": "abc123"},
|
cookies={"session": "abc123"},
|
||||||
headers={"Content-Type": "text/html"},
|
headers={"Content-Type": "text/html"},
|
||||||
request_headers={"User-Agent": "Test"},
|
request_headers={"User-Agent": "Test"},
|
||||||
encoding="utf-8"
|
encoding="utf-8",
|
||||||
)
|
)
|
||||||
|
|
||||||
assert response.url == "https://example.com"
|
assert response.url == "https://example.com"
|
||||||
@@ -250,7 +232,7 @@ class TestResponse:
|
|||||||
|
|
||||||
def test_response_with_bytes_content(self):
|
def test_response_with_bytes_content(self):
|
||||||
"""Test Response with 'bytes' content"""
|
"""Test Response with 'bytes' content"""
|
||||||
content_bytes = "<html><body>Test</body></html>".encode('utf-8')
|
content_bytes = "<html><body>Test</body></html>".encode("utf-8")
|
||||||
|
|
||||||
response = Response(
|
response = Response(
|
||||||
url="https://example.com",
|
url="https://example.com",
|
||||||
@@ -259,7 +241,7 @@ class TestResponse:
|
|||||||
reason="OK",
|
reason="OK",
|
||||||
cookies={},
|
cookies={},
|
||||||
headers={},
|
headers={},
|
||||||
request_headers={}
|
request_headers={},
|
||||||
)
|
)
|
||||||
|
|
||||||
# Should handle 'bytes' content properly
|
# Should handle 'bytes' content properly
|
||||||
@@ -268,6 +250,7 @@ class TestResponse:
|
|||||||
|
|
||||||
class _MockRequest:
|
class _MockRequest:
|
||||||
"""Minimal mock for Playwright's Request object."""
|
"""Minimal mock for Playwright's Request object."""
|
||||||
|
|
||||||
def __init__(self, url: str, resource_type: str = "document"):
|
def __init__(self, url: str, resource_type: str = "document"):
|
||||||
self.url = url
|
self.url = url
|
||||||
self.resource_type = resource_type
|
self.resource_type = resource_type
|
||||||
@@ -275,6 +258,7 @@ class _MockRequest:
|
|||||||
|
|
||||||
class _MockRoute:
|
class _MockRoute:
|
||||||
"""Minimal mock for Playwright's sync Route object."""
|
"""Minimal mock for Playwright's sync Route object."""
|
||||||
|
|
||||||
def __init__(self, url: str, resource_type: str = "document"):
|
def __init__(self, url: str, resource_type: str = "document"):
|
||||||
self.request = _MockRequest(url, resource_type)
|
self.request = _MockRequest(url, resource_type)
|
||||||
self.aborted = False
|
self.aborted = False
|
||||||
@@ -289,6 +273,7 @@ class _MockRoute:
|
|||||||
|
|
||||||
class _AsyncMockRoute:
|
class _AsyncMockRoute:
|
||||||
"""Minimal mock for Playwright's async Route object."""
|
"""Minimal mock for Playwright's async Route object."""
|
||||||
|
|
||||||
def __init__(self, url: str, resource_type: str = "document"):
|
def __init__(self, url: str, resource_type: str = "document"):
|
||||||
self.request = _MockRequest(url, resource_type)
|
self.request = _MockRequest(url, resource_type)
|
||||||
self.aborted = False
|
self.aborted = False
|
||||||
@@ -411,3 +396,90 @@ class TestCreateAsyncInterceptHandler:
|
|||||||
route = _AsyncMockRoute("https://notexample.com/page")
|
route = _AsyncMockRoute("https://notexample.com/page")
|
||||||
await handler(route)
|
await handler(route)
|
||||||
assert route.continued
|
assert route.continued
|
||||||
|
|
||||||
|
|
||||||
|
class TestIsDomainBlocked:
|
||||||
|
"""Test the frozenset-based domain matching helper."""
|
||||||
|
|
||||||
|
def test_exact_match(self):
|
||||||
|
domains = frozenset({"doubleclick.net"})
|
||||||
|
assert _is_domain_blocked("doubleclick.net", domains) is True
|
||||||
|
|
||||||
|
def test_subdomain_match(self):
|
||||||
|
domains = frozenset({"doubleclick.net"})
|
||||||
|
assert _is_domain_blocked("ads.doubleclick.net", domains) is True
|
||||||
|
|
||||||
|
def test_deep_subdomain_match(self):
|
||||||
|
domains = frozenset({"doubleclick.net"})
|
||||||
|
assert _is_domain_blocked("tracker.ads.doubleclick.net", domains) is True
|
||||||
|
|
||||||
|
def test_no_partial_match(self):
|
||||||
|
domains = frozenset({"doubleclick.net"})
|
||||||
|
assert _is_domain_blocked("notdoubleclick.net", domains) is False
|
||||||
|
|
||||||
|
def test_no_match(self):
|
||||||
|
domains = frozenset({"doubleclick.net"})
|
||||||
|
assert _is_domain_blocked("example.com", domains) is False
|
||||||
|
|
||||||
|
def test_empty_domains(self):
|
||||||
|
assert _is_domain_blocked("example.com", frozenset()) is False
|
||||||
|
|
||||||
|
def test_multiple_domains(self):
|
||||||
|
domains = frozenset({"ads.com", "tracker.io", "doubleclick.net"})
|
||||||
|
assert _is_domain_blocked("cdn.ads.com", domains) is True
|
||||||
|
assert _is_domain_blocked("tracker.io", domains) is True
|
||||||
|
assert _is_domain_blocked("safe.example.com", domains) is False
|
||||||
|
|
||||||
|
|
||||||
|
class TestAdDomains:
|
||||||
|
"""Test the built-in ad domain list."""
|
||||||
|
|
||||||
|
def test_ad_domains_is_frozenset(self):
|
||||||
|
from scrapling.engines.toolbelt.ad_domains import AD_DOMAINS
|
||||||
|
|
||||||
|
assert isinstance(AD_DOMAINS, frozenset)
|
||||||
|
|
||||||
|
def test_ad_domains_has_entries(self):
|
||||||
|
from scrapling.engines.toolbelt.ad_domains import AD_DOMAINS
|
||||||
|
|
||||||
|
assert len(AD_DOMAINS) > 1000
|
||||||
|
|
||||||
|
def test_ad_domains_contains_known_entries(self):
|
||||||
|
from scrapling.engines.toolbelt.ad_domains import AD_DOMAINS
|
||||||
|
|
||||||
|
assert "doubleclick.net" in AD_DOMAINS
|
||||||
|
assert "googlesyndication.com" in AD_DOMAINS
|
||||||
|
|
||||||
|
|
||||||
|
class TestBlockAdsConfig:
|
||||||
|
"""Test that block_ads merges ad domains into blocked_domains at config level."""
|
||||||
|
|
||||||
|
def test_block_ads_populates_blocked_domains(self):
|
||||||
|
from scrapling.engines._browsers._validators import PlaywrightConfig
|
||||||
|
|
||||||
|
config = PlaywrightConfig(block_ads=True)
|
||||||
|
assert config.blocked_domains is not None
|
||||||
|
assert len(config.blocked_domains) > 1000
|
||||||
|
assert "doubleclick.net" in config.blocked_domains
|
||||||
|
|
||||||
|
def test_block_ads_false_leaves_blocked_domains_none(self):
|
||||||
|
from scrapling.engines._browsers._validators import PlaywrightConfig
|
||||||
|
|
||||||
|
config = PlaywrightConfig(block_ads=False)
|
||||||
|
assert config.blocked_domains is None
|
||||||
|
|
||||||
|
def test_block_ads_merges_with_user_domains(self):
|
||||||
|
from scrapling.engines._browsers._validators import PlaywrightConfig
|
||||||
|
|
||||||
|
user_domains = {"my-custom-block.com"}
|
||||||
|
config = PlaywrightConfig(block_ads=True, blocked_domains=user_domains)
|
||||||
|
assert config.blocked_domains is not None
|
||||||
|
assert "my-custom-block.com" in config.blocked_domains
|
||||||
|
assert "doubleclick.net" in config.blocked_domains
|
||||||
|
|
||||||
|
def test_block_ads_does_not_modify_original_set(self):
|
||||||
|
from scrapling.engines._browsers._validators import PlaywrightConfig
|
||||||
|
|
||||||
|
user_domains = {"my-custom-block.com"}
|
||||||
|
_ = PlaywrightConfig(block_ads=True, blocked_domains=user_domains)
|
||||||
|
assert len(user_domains) == 1
|
||||||
|
|||||||
Reference in New Issue
Block a user