feat(cli): Add an option to make content safe/targets AI

This commit is contained in:
Karim shoair
2026-03-30 02:44:14 +02:00
parent 375951bd49
commit 4efbffa1dc
2 changed files with 34 additions and 10 deletions
+30 -9
View File
@@ -42,6 +42,7 @@ def __Request_and_Save(
url: str, url: str,
output_file: str, output_file: str,
css_selector: Optional[str] = None, css_selector: Optional[str] = None,
ai_targeted: bool = False,
**kwargs, **kwargs,
) -> None: ) -> None:
"""Make a request using the specified fetcher function and save the result""" """Make a request using the specified fetcher function and save the result"""
@@ -53,7 +54,7 @@ def __Request_and_Save(
output_path = Path.cwd() / output_file output_path = Path.cwd() / output_file
response = fetcher_func(url, **kwargs) response = fetcher_func(url, **kwargs)
Convertor.write_content_to_file(response, str(output_path), css_selector) Convertor.write_content_to_file(response, str(output_path), css_selector, main_content_only=ai_targeted)
log.info(f"Content successfully saved to '{output_path}'") log.info(f"Content successfully saved to '{output_path}'")
@@ -202,6 +203,12 @@ def extract():
def _common_http_options(f): def _common_http_options(f):
"""Apply shared Click options for all HTTP extract commands (get/post/put/delete).""" """Apply shared Click options for all HTTP extract commands (get/post/put/delete)."""
decorators = [ decorators = [
option(
"--ai-targeted",
is_flag=True,
default=False,
help="Extract only main content and sanitize hidden elements for AI consumption (default: False)",
),
option( option(
"--stealthy-headers/--no-stealthy-headers", "--stealthy-headers/--no-stealthy-headers",
default=True, default=True,
@@ -250,6 +257,12 @@ def _common_http_options(f):
def _common_browser_options(f): def _common_browser_options(f):
"""Apply shared Click options for browser-based commands (fetch/stealthy_fetch).""" """Apply shared Click options for browser-based commands (fetch/stealthy_fetch)."""
decorators = [ decorators = [
option(
"--ai-targeted",
is_flag=True,
default=False,
help="Extract only main content and sanitize hidden elements for AI consumption (default: False)",
),
option( option(
"--extra-headers", "--extra-headers",
"-H", "-H",
@@ -317,11 +330,13 @@ def _data_options(f):
return f return f
def __http_command(method_name: str, url: str, output_file: str, css_selector: Optional[str], **kwargs) -> None: def __http_command(
method_name: str, url: str, output_file: str, css_selector: Optional[str], ai_targeted: bool = False, **kwargs
) -> None:
"""Shared implementation for HTTP extract commands.""" """Shared implementation for HTTP extract commands."""
from scrapling.fetchers import Fetcher from scrapling.fetchers import Fetcher
__Request_and_Save(getattr(Fetcher, method_name), url, output_file, css_selector, **kwargs) __Request_and_Save(getattr(Fetcher, method_name), url, output_file, css_selector, ai_targeted=ai_targeted, **kwargs)
@extract.command(help=f"Perform a GET request and save the content to a file.\n\n{__OUTPUT_FILE_HELP__}") @extract.command(help=f"Perform a GET request and save the content to a file.\n\n{__OUTPUT_FILE_HELP__}")
@@ -341,6 +356,7 @@ def get(
verify, verify,
impersonate, impersonate,
stealthy_headers, stealthy_headers,
ai_targeted,
): ):
"""Perform a GET request and save the content to a file.""" """Perform a GET request and save the content to a file."""
kwargs = __BuildRequest( kwargs = __BuildRequest(
@@ -355,7 +371,7 @@ def get(
impersonate=impersonate, impersonate=impersonate,
proxy=proxy, proxy=proxy,
) )
__http_command("get", url, output_file, css_selector, **kwargs) __http_command("get", url, output_file, css_selector, ai_targeted=ai_targeted, **kwargs)
@extract.command(help=f"Perform a POST request and save the content to a file.\n\n{__OUTPUT_FILE_HELP__}") @extract.command(help=f"Perform a POST request and save the content to a file.\n\n{__OUTPUT_FILE_HELP__}")
@@ -378,6 +394,7 @@ def post(
verify, verify,
impersonate, impersonate,
stealthy_headers, stealthy_headers,
ai_targeted,
): ):
"""Perform a POST request and save the content to a file.""" """Perform a POST request and save the content to a file."""
kwargs = __BuildRequest( kwargs = __BuildRequest(
@@ -393,7 +410,7 @@ def post(
proxy=proxy, proxy=proxy,
data=data, data=data,
) )
__http_command("post", url, output_file, css_selector, **kwargs) __http_command("post", url, output_file, css_selector, ai_targeted=ai_targeted, **kwargs)
@extract.command(help=f"Perform a PUT request and save the content to a file.\n\n{__OUTPUT_FILE_HELP__}") @extract.command(help=f"Perform a PUT request and save the content to a file.\n\n{__OUTPUT_FILE_HELP__}")
@@ -416,6 +433,7 @@ def put(
verify, verify,
impersonate, impersonate,
stealthy_headers, stealthy_headers,
ai_targeted,
): ):
"""Perform a PUT request and save the content to a file.""" """Perform a PUT request and save the content to a file."""
kwargs = __BuildRequest( kwargs = __BuildRequest(
@@ -431,7 +449,7 @@ def put(
proxy=proxy, proxy=proxy,
data=data, data=data,
) )
__http_command("put", url, output_file, css_selector, **kwargs) __http_command("put", url, output_file, css_selector, ai_targeted=ai_targeted, **kwargs)
@extract.command(help=f"Perform a DELETE request and save the content to a file.\n\n{__OUTPUT_FILE_HELP__}") @extract.command(help=f"Perform a DELETE request and save the content to a file.\n\n{__OUTPUT_FILE_HELP__}")
@@ -451,6 +469,7 @@ def delete(
verify, verify,
impersonate, impersonate,
stealthy_headers, stealthy_headers,
ai_targeted,
): ):
"""Perform a DELETE request and save the content to a file.""" """Perform a DELETE request and save the content to a file."""
kwargs = __BuildRequest( kwargs = __BuildRequest(
@@ -465,7 +484,7 @@ def delete(
impersonate=impersonate, impersonate=impersonate,
proxy=proxy, proxy=proxy,
) )
__http_command("delete", url, output_file, css_selector, **kwargs) __http_command("delete", url, output_file, css_selector, ai_targeted=ai_targeted, **kwargs)
def __build_browser_kwargs( def __build_browser_kwargs(
@@ -518,6 +537,7 @@ def fetch(
real_chrome, real_chrome,
proxy, proxy,
extra_headers, extra_headers,
ai_targeted,
): ):
"""Opens up a browser and fetch content using DynamicFetcher.""" """Opens up a browser and fetch content using DynamicFetcher."""
parsed_headers, _ = _ParseHeaders(extra_headers, False) parsed_headers, _ = _ParseHeaders(extra_headers, False)
@@ -535,7 +555,7 @@ def fetch(
) )
from scrapling.fetchers import DynamicFetcher from scrapling.fetchers import DynamicFetcher
__Request_and_Save(DynamicFetcher.fetch, url, output_file, css_selector, **kwargs) __Request_and_Save(DynamicFetcher.fetch, url, output_file, css_selector, ai_targeted=ai_targeted, **kwargs)
@extract.command(help=f"Use StealthyFetcher to fetch content with advanced stealth features.\n\n{__OUTPUT_FILE_HELP__}") @extract.command(help=f"Use StealthyFetcher to fetch content with advanced stealth features.\n\n{__OUTPUT_FILE_HELP__}")
@@ -576,6 +596,7 @@ def stealthy_fetch(
solve_cloudflare, solve_cloudflare,
allow_webgl, allow_webgl,
hide_canvas, hide_canvas,
ai_targeted,
): ):
"""Opens up a browser with advanced stealth features and fetch content using StealthyFetcher.""" """Opens up a browser with advanced stealth features and fetch content using StealthyFetcher."""
parsed_headers, _ = _ParseHeaders(extra_headers, False) parsed_headers, _ = _ParseHeaders(extra_headers, False)
@@ -601,7 +622,7 @@ def stealthy_fetch(
) )
from scrapling.fetchers import StealthyFetcher from scrapling.fetchers import StealthyFetcher
__Request_and_Save(StealthyFetcher.fetch, url, output_file, css_selector, **kwargs) __Request_and_Save(StealthyFetcher.fetch, url, output_file, css_selector, ai_targeted=ai_targeted, **kwargs)
@group() @group()
+4 -1
View File
@@ -653,7 +653,9 @@ class Convertor:
yield "" yield ""
@classmethod @classmethod
def write_content_to_file(cls, page: Selector, filename: str, css_selector: Optional[str] = None) -> None: def write_content_to_file(
cls, page: Selector, filename: str, css_selector: Optional[str] = None, main_content_only: bool = False
) -> None:
"""Write a Selector's content to a file""" """Write a Selector's content to a file"""
if not page or not isinstance(page, Selector): # pragma: no cover if not page or not isinstance(page, Selector): # pragma: no cover
raise TypeError("Input must be of type `Selector`") raise TypeError("Input must be of type `Selector`")
@@ -670,6 +672,7 @@ class Convertor:
page, page,
cls._extension_map[extension], cls._extension_map[extension],
css_selector=css_selector, css_selector=css_selector,
main_content_only=main_content_only,
) )
) )
) )