feat(extract): Adding new command to CLI options + Optimizations
Users can now fetch websites directly without code and extract full/selected HTML content as HTML, Markdown, or extract text content.
This commit is contained in:
+785
-9
@@ -1,21 +1,75 @@
|
||||
import os
|
||||
import sys
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from subprocess import check_output
|
||||
from sys import executable as python_executable
|
||||
|
||||
from click import command, option, Choice, group
|
||||
from scrapling.core.utils import log
|
||||
from scrapling.core.shell import Convertor, _CookieParser
|
||||
from scrapling.fetchers import Fetcher, DynamicFetcher, StealthyFetcher
|
||||
|
||||
from orjson import loads as json_loads, JSONDecodeError
|
||||
from click import command, option, Choice, group, argument
|
||||
|
||||
__OUTPUT_FILE_HELP__ = "Output file path can be HTML content, Markdown of the HTML content, or the text content. Use file extensions (`.html`/`.md`/`.txt`) respectively."
|
||||
|
||||
|
||||
def get_package_dir():
|
||||
return Path(os.path.dirname(__file__))
|
||||
return Path(__file__).parent
|
||||
|
||||
|
||||
def run_command(cmd, line):
|
||||
print(f"Installing {line}...")
|
||||
_ = subprocess.check_call(cmd, shell=False) # nosec B603
|
||||
_ = check_output(cmd, shell=False) # nosec B603
|
||||
# I meant to not use try except here
|
||||
|
||||
|
||||
def parse_headers(header_strings):
|
||||
"""Parse header strings into a dictionary"""
|
||||
headers = {}
|
||||
for header in header_strings:
|
||||
if ":" in header:
|
||||
key, value = header.split(":", 1)
|
||||
headers[key.strip()] = value.strip()
|
||||
else:
|
||||
log.warning(f"Invalid header format '{header}', should be 'Key: Value'")
|
||||
return headers
|
||||
|
||||
|
||||
def parse_cookies(cookie_string):
|
||||
"""Parse cookie string into a dictionary"""
|
||||
if not cookie_string:
|
||||
return {}
|
||||
|
||||
try:
|
||||
cookies = {key: value for key, value in _CookieParser(cookie_string)}
|
||||
except Exception as e:
|
||||
raise ValueError(f"Could not parse cookies '{cookie_string}': {e}")
|
||||
|
||||
return cookies
|
||||
|
||||
|
||||
def parse_json_data(json_string):
|
||||
"""Parse JSON string into a Python object"""
|
||||
if not json_string:
|
||||
return None
|
||||
|
||||
try:
|
||||
return json_loads(json_string)
|
||||
except JSONDecodeError as e:
|
||||
raise ValueError(f"Invalid JSON data '{json_string}': {e}")
|
||||
|
||||
|
||||
def make_request_and_save(fetcher_func, url, output_file, css_selector=None, **kwargs):
|
||||
"""Make a request using the specified fetcher function and save the result"""
|
||||
# Handle relative paths - convert to an absolute path based on the current working directory
|
||||
output_path = Path(output_file)
|
||||
if not output_path.is_absolute():
|
||||
output_path = Path.cwd() / output_file
|
||||
|
||||
response = fetcher_func(url, **kwargs)
|
||||
Convertor.write_content_to_file(response, str(output_path), css_selector)
|
||||
log.info(f"Content successfully saved to '{output_path}'")
|
||||
|
||||
|
||||
@command(help="Install all Scrapling's Fetchers dependencies")
|
||||
@option(
|
||||
"-f",
|
||||
@@ -32,15 +86,22 @@ def install(force):
|
||||
or not get_package_dir().joinpath(".scrapling_dependencies_installed").exists()
|
||||
):
|
||||
run_command(
|
||||
[sys.executable, "-m", "playwright", "install", "chromium"],
|
||||
[python_executable, "-m", "playwright", "install", "chromium"],
|
||||
"Playwright browsers",
|
||||
)
|
||||
run_command(
|
||||
[sys.executable, "-m", "playwright", "install-deps", "chromium", "firefox"],
|
||||
[
|
||||
python_executable,
|
||||
"-m",
|
||||
"playwright",
|
||||
"install-deps",
|
||||
"chromium",
|
||||
"firefox",
|
||||
],
|
||||
"Playwright dependencies",
|
||||
)
|
||||
run_command(
|
||||
[sys.executable, "-m", "camoufox", "fetch", "--browserforge"],
|
||||
[python_executable, "-m", "camoufox", "fetch", "--browserforge"],
|
||||
"Camoufox browser and databases",
|
||||
)
|
||||
# if no errors raised by the above commands, then we add the below file
|
||||
@@ -77,6 +138,720 @@ def shell(code, level):
|
||||
console.start()
|
||||
|
||||
|
||||
def parse_extract_arguments(headers, cookies, params, json=None):
|
||||
"""Parse arguments for extract command"""
|
||||
parsed_headers = parse_headers(headers)
|
||||
parsed_cookies = parse_cookies(cookies)
|
||||
parsed_json = parse_json_data(json)
|
||||
parsed_params = {}
|
||||
for param in params:
|
||||
if "=" in param:
|
||||
key, value = param.split("=", 1)
|
||||
parsed_params[key] = value
|
||||
|
||||
return parsed_headers, parsed_cookies, parsed_params, parsed_json
|
||||
|
||||
|
||||
@group(
|
||||
help="Fetch web pages using various fetchers and extract full/selected HTML content as HTML, Markdown, or extract text content."
|
||||
)
|
||||
def extract():
|
||||
"""Extract content from web pages and save to files"""
|
||||
pass
|
||||
|
||||
|
||||
@extract.command(
|
||||
help=f"Perform a GET request and save content to file.\n\n{__OUTPUT_FILE_HELP__}"
|
||||
)
|
||||
@argument("url", required=True)
|
||||
@argument("output_file", required=True)
|
||||
@option(
|
||||
"--headers",
|
||||
"-H",
|
||||
multiple=True,
|
||||
help='HTTP headers in format "Key: Value" (can be used multiple times)',
|
||||
)
|
||||
@option("--cookies", help='Cookies string in format "name1=value1; name2=value2"')
|
||||
@option(
|
||||
"--timeout", type=int, default=30, help="Request timeout in seconds (default: 30)"
|
||||
)
|
||||
@option("--proxy", help='Proxy URL in format "http://username:password@host:port"')
|
||||
@option(
|
||||
"--css-selector",
|
||||
"-s",
|
||||
help="CSS selector to extract specific content from the page. It resolves to the first match if multiple matches are found.",
|
||||
)
|
||||
@option(
|
||||
"--params",
|
||||
"-p",
|
||||
multiple=True,
|
||||
help='Query parameters in format "key=value" (can be used multiple times)',
|
||||
)
|
||||
@option(
|
||||
"--follow-redirects/--no-follow-redirects",
|
||||
default=True,
|
||||
help="Whether to follow redirects (default: True)",
|
||||
)
|
||||
@option(
|
||||
"--verify/--no-verify",
|
||||
default=True,
|
||||
help="Whether to verify SSL certificates (default: True)",
|
||||
)
|
||||
@option("--impersonate", help="Browser to impersonate (e.g., chrome, firefox).")
|
||||
@option(
|
||||
"--stealthy-headers/--no-stealthy-headers",
|
||||
default=True,
|
||||
help="Use stealthy browser headers (default: True)",
|
||||
)
|
||||
def get(
|
||||
url,
|
||||
output_file,
|
||||
headers,
|
||||
cookies,
|
||||
timeout,
|
||||
proxy,
|
||||
css_selector,
|
||||
params,
|
||||
follow_redirects,
|
||||
verify,
|
||||
impersonate,
|
||||
stealthy_headers,
|
||||
):
|
||||
"""
|
||||
Perform a GET request and save content to file.
|
||||
|
||||
:param url: Target URL for the request.
|
||||
:param output_file: Output file path (.md for Markdown, .html for HTML).
|
||||
:param headers: HTTP headers to include in the request.
|
||||
:param cookies: Cookies to use in the request.
|
||||
:param timeout: Number of seconds to wait before timing out.
|
||||
:param proxy: Proxy URL to use. (Format: "http://username:password@localhost:8030")
|
||||
:param css_selector: CSS selector to extract specific content.
|
||||
:param params: Query string parameters for the request.
|
||||
:param follow_redirects: Whether to follow redirects.
|
||||
:param verify: Whether to verify HTTPS certificates.
|
||||
:param impersonate: Browser version to impersonate.
|
||||
:param stealthy_headers: If enabled, creates and adds real browser headers.
|
||||
"""
|
||||
|
||||
# Parse parameters
|
||||
parsed_headers, parsed_cookies, parsed_params, _ = parse_extract_arguments(
|
||||
headers, cookies, params
|
||||
)
|
||||
|
||||
# Build request arguments
|
||||
kwargs = {
|
||||
"headers": parsed_headers if parsed_headers else None,
|
||||
"cookies": parsed_cookies if parsed_cookies else None,
|
||||
"timeout": timeout,
|
||||
"follow_redirects": follow_redirects,
|
||||
"verify": verify,
|
||||
"stealthy_headers": stealthy_headers,
|
||||
"impersonate": impersonate,
|
||||
}
|
||||
|
||||
if parsed_params:
|
||||
kwargs["params"] = parsed_params
|
||||
if proxy:
|
||||
kwargs["proxy"] = proxy
|
||||
|
||||
make_request_and_save(Fetcher.get, url, output_file, css_selector, **kwargs)
|
||||
|
||||
|
||||
@extract.command(
|
||||
help=f"Perform a POST request and save content to file.\n\n{__OUTPUT_FILE_HELP__}"
|
||||
)
|
||||
@argument("url", required=True)
|
||||
@argument("output_file", required=True)
|
||||
@option(
|
||||
"--data",
|
||||
"-d",
|
||||
help='Form data to include in the request body (as string, ex: "param1=value1¶m2=value2")',
|
||||
)
|
||||
@option("--json", "-j", help="JSON data to include in the request body (as string)")
|
||||
@option(
|
||||
"--headers",
|
||||
"-H",
|
||||
multiple=True,
|
||||
help='HTTP headers in format "Key: Value" (can be used multiple times)',
|
||||
)
|
||||
@option("--cookies", help='Cookies string in format "name1=value1; name2=value2"')
|
||||
@option(
|
||||
"--timeout", type=int, default=30, help="Request timeout in seconds (default: 30)"
|
||||
)
|
||||
@option("--proxy", help='Proxy URL in format "http://username:password@host:port"')
|
||||
@option(
|
||||
"--css-selector",
|
||||
"-s",
|
||||
help="CSS selector to extract specific content from the page",
|
||||
)
|
||||
@option(
|
||||
"--params",
|
||||
"-p",
|
||||
multiple=True,
|
||||
help='Query parameters in format "key=value" (can be used multiple times)',
|
||||
)
|
||||
@option(
|
||||
"--follow-redirects/--no-follow-redirects",
|
||||
default=True,
|
||||
help="Whether to follow redirects (default: True)",
|
||||
)
|
||||
@option(
|
||||
"--verify/--no-verify",
|
||||
default=True,
|
||||
help="Whether to verify SSL certificates (default: True)",
|
||||
)
|
||||
@option("--impersonate", help="Browser to impersonate (e.g., chrome, firefox).")
|
||||
@option(
|
||||
"--stealthy-headers/--no-stealthy-headers",
|
||||
default=True,
|
||||
help="Use stealthy browser headers (default: True)",
|
||||
)
|
||||
def post(
|
||||
url,
|
||||
output_file,
|
||||
data,
|
||||
json,
|
||||
headers,
|
||||
cookies,
|
||||
timeout,
|
||||
proxy,
|
||||
css_selector,
|
||||
params,
|
||||
follow_redirects,
|
||||
verify,
|
||||
impersonate,
|
||||
stealthy_headers,
|
||||
):
|
||||
"""
|
||||
Perform a POST request and save content to file.
|
||||
|
||||
:param url: Target URL for the request.
|
||||
:param output_file: Output file path (.md for Markdown, .html for HTML).
|
||||
:param data: Form data to include in the request body. (as string, ex: "param1=value1¶m2=value2")
|
||||
:param json: A JSON serializable object to include in the body of the request.
|
||||
:param headers: Headers to include in the request.
|
||||
:param cookies: Cookies to use in the request.
|
||||
:param timeout: Number of seconds to wait before timing out.
|
||||
:param proxy: Proxy URL to use.
|
||||
:param css_selector: CSS selector to extract specific content.
|
||||
:param params: Query string parameters for the request.
|
||||
:param follow_redirects: Whether to follow redirects.
|
||||
:param verify: Whether to verify HTTPS certificates.
|
||||
:param impersonate: Browser version to impersonate.
|
||||
:param stealthy_headers: If enabled, creates and adds real browser headers.
|
||||
"""
|
||||
|
||||
# Parse parameters
|
||||
parsed_headers, parsed_cookies, parsed_params, parsed_json = (
|
||||
parse_extract_arguments(headers, cookies, params, json)
|
||||
)
|
||||
|
||||
# Build request arguments
|
||||
kwargs = {
|
||||
"headers": parsed_headers if parsed_headers else None,
|
||||
"cookies": parsed_cookies if parsed_cookies else None,
|
||||
"timeout": timeout,
|
||||
"follow_redirects": follow_redirects,
|
||||
"verify": verify,
|
||||
"stealthy_headers": stealthy_headers,
|
||||
"impersonate": impersonate,
|
||||
}
|
||||
|
||||
if data:
|
||||
kwargs["data"] = data
|
||||
if parsed_json:
|
||||
kwargs["json"] = parsed_json
|
||||
if parsed_params:
|
||||
kwargs["params"] = parsed_params
|
||||
if proxy:
|
||||
kwargs["proxy"] = proxy
|
||||
|
||||
make_request_and_save(Fetcher.post, url, output_file, css_selector, **kwargs)
|
||||
|
||||
|
||||
@extract.command(
|
||||
help=f"Perform a PUT request and save content to file.\n\n{__OUTPUT_FILE_HELP__}"
|
||||
)
|
||||
@argument("url", required=True)
|
||||
@argument("output_file", required=True)
|
||||
@option("--data", "-d", help="Form data to include in the request body")
|
||||
@option("--json", "-j", help="JSON data to include in the request body (as string)")
|
||||
@option(
|
||||
"--headers",
|
||||
"-H",
|
||||
multiple=True,
|
||||
help='HTTP headers in format "Key: Value" (can be used multiple times)',
|
||||
)
|
||||
@option("--cookies", help='Cookies string in format "name1=value1; name2=value2"')
|
||||
@option(
|
||||
"--timeout", type=int, default=30, help="Request timeout in seconds (default: 30)"
|
||||
)
|
||||
@option("--proxy", help='Proxy URL in format "http://username:password@host:port"')
|
||||
@option(
|
||||
"--css-selector",
|
||||
"-s",
|
||||
help="CSS selector to extract specific content from the page",
|
||||
)
|
||||
@option(
|
||||
"--params",
|
||||
"-p",
|
||||
multiple=True,
|
||||
help='Query parameters in format "key=value" (can be used multiple times)',
|
||||
)
|
||||
@option(
|
||||
"--follow-redirects/--no-follow-redirects",
|
||||
default=True,
|
||||
help="Whether to follow redirects (default: True)",
|
||||
)
|
||||
@option(
|
||||
"--verify/--no-verify",
|
||||
default=True,
|
||||
help="Whether to verify SSL certificates (default: True)",
|
||||
)
|
||||
@option("--impersonate", help="Browser to impersonate (e.g., chrome, firefox).")
|
||||
@option(
|
||||
"--stealthy-headers/--no-stealthy-headers",
|
||||
default=True,
|
||||
help="Use stealthy browser headers (default: True)",
|
||||
)
|
||||
def put(
|
||||
url,
|
||||
output_file,
|
||||
data,
|
||||
json,
|
||||
headers,
|
||||
cookies,
|
||||
timeout,
|
||||
proxy,
|
||||
css_selector,
|
||||
params,
|
||||
follow_redirects,
|
||||
verify,
|
||||
impersonate,
|
||||
stealthy_headers,
|
||||
):
|
||||
"""
|
||||
Perform a PUT request and save content to file.
|
||||
|
||||
:param url: Target URL for the request.
|
||||
:param output_file: Output file path (.md for Markdown, .html for HTML).
|
||||
:param data: Form data to include in the request body.
|
||||
:param json: A JSON serializable object to include in the body of the request.
|
||||
:param headers: Headers to include in the request.
|
||||
:param cookies: Cookies to use in the request.
|
||||
:param timeout: Number of seconds to wait before timing out.
|
||||
:param proxy: Proxy URL to use.
|
||||
:param css_selector: CSS selector to extract specific content.
|
||||
:param params: Query string parameters for the request.
|
||||
:param follow_redirects: Whether to follow redirects.
|
||||
:param verify: Whether to verify HTTPS certificates.
|
||||
:param impersonate: Browser version to impersonate.
|
||||
:param stealthy_headers: If enabled, creates and adds real browser headers.
|
||||
"""
|
||||
|
||||
# Parse parameters
|
||||
parsed_headers, parsed_cookies, parsed_params, parsed_json = (
|
||||
parse_extract_arguments(headers, cookies, params, json)
|
||||
)
|
||||
|
||||
# Build request arguments
|
||||
kwargs = {
|
||||
"headers": parsed_headers if parsed_headers else None,
|
||||
"cookies": parsed_cookies if parsed_cookies else None,
|
||||
"timeout": timeout,
|
||||
"follow_redirects": follow_redirects,
|
||||
"verify": verify,
|
||||
"stealthy_headers": stealthy_headers,
|
||||
"impersonate": impersonate,
|
||||
}
|
||||
|
||||
if data:
|
||||
kwargs["data"] = data
|
||||
if parsed_json:
|
||||
kwargs["json"] = parsed_json
|
||||
if parsed_params:
|
||||
kwargs["params"] = parsed_params
|
||||
if proxy:
|
||||
kwargs["proxy"] = proxy
|
||||
|
||||
make_request_and_save(Fetcher.put, url, output_file, css_selector, **kwargs)
|
||||
|
||||
|
||||
@extract.command(
|
||||
help=f"Perform a DELETE request and save content to file.\n\n{__OUTPUT_FILE_HELP__}"
|
||||
)
|
||||
@argument("url", required=True)
|
||||
@argument("output_file", required=True)
|
||||
@option(
|
||||
"--headers",
|
||||
"-H",
|
||||
multiple=True,
|
||||
help='HTTP headers in format "Key: Value" (can be used multiple times)',
|
||||
)
|
||||
@option("--cookies", help='Cookies string in format "name1=value1; name2=value2"')
|
||||
@option(
|
||||
"--timeout", type=int, default=30, help="Request timeout in seconds (default: 30)"
|
||||
)
|
||||
@option("--proxy", help='Proxy URL in format "http://username:password@host:port"')
|
||||
@option(
|
||||
"--css-selector",
|
||||
"-s",
|
||||
help="CSS selector to extract specific content from the page",
|
||||
)
|
||||
@option(
|
||||
"--params",
|
||||
"-p",
|
||||
multiple=True,
|
||||
help='Query parameters in format "key=value" (can be used multiple times)',
|
||||
)
|
||||
@option(
|
||||
"--follow-redirects/--no-follow-redirects",
|
||||
default=True,
|
||||
help="Whether to follow redirects (default: True)",
|
||||
)
|
||||
@option(
|
||||
"--verify/--no-verify",
|
||||
default=True,
|
||||
help="Whether to verify SSL certificates (default: True)",
|
||||
)
|
||||
@option("--impersonate", help="Browser to impersonate (e.g., chrome, firefox).")
|
||||
@option(
|
||||
"--stealthy-headers/--no-stealthy-headers",
|
||||
default=True,
|
||||
help="Use stealthy browser headers (default: True)",
|
||||
)
|
||||
def delete(
|
||||
url,
|
||||
output_file,
|
||||
headers,
|
||||
cookies,
|
||||
timeout,
|
||||
proxy,
|
||||
css_selector,
|
||||
params,
|
||||
follow_redirects,
|
||||
verify,
|
||||
impersonate,
|
||||
stealthy_headers,
|
||||
):
|
||||
"""
|
||||
Perform a DELETE request and save content to file.
|
||||
|
||||
:param url: Target URL for the request.
|
||||
:param output_file: Output file path (.md for Markdown, .html for HTML).
|
||||
:param headers: Headers to include in the request.
|
||||
:param cookies: Cookies to use in the request.
|
||||
:param timeout: Number of seconds to wait before timing out.
|
||||
:param proxy: Proxy URL to use.
|
||||
:param css_selector: CSS selector to extract specific content.
|
||||
:param params: Query string parameters for the request.
|
||||
:param follow_redirects: Whether to follow redirects.
|
||||
:param verify: Whether to verify HTTPS certificates.
|
||||
:param impersonate: Browser version to impersonate.
|
||||
:param stealthy_headers: If enabled, creates and adds real browser headers.
|
||||
"""
|
||||
|
||||
# Parse parameters
|
||||
parsed_headers, parsed_cookies, parsed_params, _ = parse_extract_arguments(
|
||||
headers, cookies, params
|
||||
)
|
||||
|
||||
# Build request arguments
|
||||
kwargs = {
|
||||
"headers": parsed_headers if parsed_headers else None,
|
||||
"cookies": parsed_cookies if parsed_cookies else None,
|
||||
"timeout": timeout,
|
||||
"follow_redirects": follow_redirects,
|
||||
"verify": verify,
|
||||
"stealthy_headers": stealthy_headers,
|
||||
"impersonate": impersonate,
|
||||
}
|
||||
|
||||
if parsed_params:
|
||||
kwargs["params"] = parsed_params
|
||||
if proxy:
|
||||
kwargs["proxy"] = proxy
|
||||
|
||||
make_request_and_save(Fetcher.delete, url, output_file, css_selector, **kwargs)
|
||||
|
||||
|
||||
@extract.command(
|
||||
help=f"Use DynamicFetcher to fetch content with browser automation.\n\n{__OUTPUT_FILE_HELP__}"
|
||||
)
|
||||
@argument("url", required=True)
|
||||
@argument("output_file", required=True)
|
||||
@option(
|
||||
"--headless/--no-headless",
|
||||
default=True,
|
||||
help="Run browser in headless mode (default: True)",
|
||||
)
|
||||
@option(
|
||||
"--disable-resources/--enable-resources",
|
||||
default=False,
|
||||
help="Drop unnecessary resources for speed boost (default: False)",
|
||||
)
|
||||
@option(
|
||||
"--network-idle/--no-network-idle",
|
||||
default=False,
|
||||
help="Wait for network idle (default: False)",
|
||||
)
|
||||
@option(
|
||||
"--timeout",
|
||||
type=int,
|
||||
default=30000,
|
||||
help="Timeout in milliseconds (default: 30000)",
|
||||
)
|
||||
@option(
|
||||
"--wait",
|
||||
type=int,
|
||||
default=0,
|
||||
help="Additional wait time in milliseconds after page load (default: 0)",
|
||||
)
|
||||
@option(
|
||||
"--css-selector",
|
||||
"-s",
|
||||
help="CSS selector to extract specific content from the page",
|
||||
)
|
||||
@option("--wait-selector", help="CSS selector to wait for before proceeding")
|
||||
@option("--locale", default="en-US", help="Browser locale (default: en-US)")
|
||||
@option(
|
||||
"--stealth/--no-stealth", default=False, help="Enable stealth mode (default: False)"
|
||||
)
|
||||
@option(
|
||||
"--hide-canvas/--show-canvas",
|
||||
default=False,
|
||||
help="Add noise to canvas operations (default: False)",
|
||||
)
|
||||
@option(
|
||||
"--disable-webgl/--enable-webgl",
|
||||
default=False,
|
||||
help="Disable WebGL support (default: False)",
|
||||
)
|
||||
@option("--proxy", help='Proxy URL in format "http://username:password@host:port"')
|
||||
@option(
|
||||
"--extra-headers",
|
||||
"-H",
|
||||
multiple=True,
|
||||
help='Extra headers in format "Key: Value" (can be used multiple times)',
|
||||
)
|
||||
def fetch(
|
||||
url,
|
||||
output_file,
|
||||
headless,
|
||||
disable_resources,
|
||||
network_idle,
|
||||
timeout,
|
||||
wait,
|
||||
css_selector,
|
||||
wait_selector,
|
||||
locale,
|
||||
stealth,
|
||||
hide_canvas,
|
||||
disable_webgl,
|
||||
proxy,
|
||||
extra_headers,
|
||||
):
|
||||
"""
|
||||
Opens up a browser and fetch content using DynamicFetcher.
|
||||
|
||||
:param url: Target url.
|
||||
:param output_file: Output file path (.md for Markdown, .html for HTML).
|
||||
:param headless: Run the browser in headless/hidden or headful/visible mode.
|
||||
:param disable_resources: Drop requests of unnecessary resources for a speed boost.
|
||||
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
||||
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page.
|
||||
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before returning.
|
||||
:param css_selector: CSS selector to extract specific content.
|
||||
:param wait_selector: Wait for a specific CSS selector to be in a specific state.
|
||||
:param locale: Set the locale for the browser.
|
||||
:param stealth: Enables stealth mode.
|
||||
:param hide_canvas: Add random noise to canvas operations to prevent fingerprinting.
|
||||
:param disable_webgl: Disables WebGL and WebGL 2.0 support entirely.
|
||||
:param proxy: The proxy to be used with requests.
|
||||
:param extra_headers: Extra headers to add to the request.
|
||||
"""
|
||||
|
||||
# Parse parameters
|
||||
parsed_headers = parse_headers(extra_headers)
|
||||
|
||||
# Build request arguments
|
||||
kwargs = {
|
||||
"headless": headless,
|
||||
"disable_resources": disable_resources,
|
||||
"network_idle": network_idle,
|
||||
"timeout": timeout,
|
||||
"locale": locale,
|
||||
"stealth": stealth,
|
||||
"hide_canvas": hide_canvas,
|
||||
"disable_webgl": disable_webgl,
|
||||
}
|
||||
|
||||
if wait > 0:
|
||||
kwargs["wait"] = wait
|
||||
if wait_selector:
|
||||
kwargs["wait_selector"] = wait_selector
|
||||
if proxy:
|
||||
kwargs["proxy"] = proxy
|
||||
if parsed_headers:
|
||||
kwargs["extra_headers"] = parsed_headers
|
||||
|
||||
make_request_and_save(
|
||||
DynamicFetcher.fetch, url, output_file, css_selector, **kwargs
|
||||
)
|
||||
|
||||
|
||||
@extract.command(
|
||||
help=f"Use StealthyFetcher to fetch content with advanced stealth features.\n\n{__OUTPUT_FILE_HELP__}"
|
||||
)
|
||||
@argument("url", required=True)
|
||||
@argument("output_file", required=True)
|
||||
@option(
|
||||
"--headless/--no-headless",
|
||||
default=True,
|
||||
help="Run browser in headless mode (default: True)",
|
||||
)
|
||||
@option(
|
||||
"--block-images/--allow-images",
|
||||
default=False,
|
||||
help="Block image loading (default: False)",
|
||||
)
|
||||
@option(
|
||||
"--disable-resources/--enable-resources",
|
||||
default=False,
|
||||
help="Drop unnecessary resources for speed boost (default: False)",
|
||||
)
|
||||
@option(
|
||||
"--block-webrtc/--allow-webrtc",
|
||||
default=False,
|
||||
help="Block WebRTC entirely (default: False)",
|
||||
)
|
||||
@option(
|
||||
"--humanize/--no-humanize",
|
||||
default=False,
|
||||
help="Humanize cursor movement (default: False)",
|
||||
)
|
||||
@option(
|
||||
"--solve-cloudflare/--no-solve-cloudflare",
|
||||
default=False,
|
||||
help="Solve Cloudflare challenges (default: False)",
|
||||
)
|
||||
@option("--allow-webgl/--block-webgl", default=True, help="Allow WebGL (default: True)")
|
||||
@option(
|
||||
"--network-idle/--no-network-idle",
|
||||
default=False,
|
||||
help="Wait for network idle (default: False)",
|
||||
)
|
||||
@option(
|
||||
"--disable-ads/--allow-ads",
|
||||
default=False,
|
||||
help="Install uBlock Origin addon (default: False)",
|
||||
)
|
||||
@option(
|
||||
"--timeout",
|
||||
type=int,
|
||||
default=30000,
|
||||
help="Timeout in milliseconds (default: 30000)",
|
||||
)
|
||||
@option(
|
||||
"--wait",
|
||||
type=int,
|
||||
default=0,
|
||||
help="Additional wait time in milliseconds after page load (default: 0)",
|
||||
)
|
||||
@option(
|
||||
"--css-selector",
|
||||
"-s",
|
||||
help="CSS selector to extract specific content from the page",
|
||||
)
|
||||
@option("--wait-selector", help="CSS selector to wait for before proceeding")
|
||||
@option(
|
||||
"--geoip/--no-geoip",
|
||||
default=False,
|
||||
help="Use IP geolocation for timezone/locale (default: False)",
|
||||
)
|
||||
@option("--proxy", help='Proxy URL in format "http://username:password@host:port"')
|
||||
@option(
|
||||
"--extra-headers",
|
||||
"-H",
|
||||
multiple=True,
|
||||
help='Extra headers in format "Key: Value" (can be used multiple times)',
|
||||
)
|
||||
def stealthy_fetch(
|
||||
url,
|
||||
output_file,
|
||||
headless,
|
||||
block_images,
|
||||
disable_resources,
|
||||
block_webrtc,
|
||||
humanize,
|
||||
solve_cloudflare,
|
||||
allow_webgl,
|
||||
network_idle,
|
||||
disable_ads,
|
||||
timeout,
|
||||
wait,
|
||||
css_selector,
|
||||
wait_selector,
|
||||
geoip,
|
||||
proxy,
|
||||
extra_headers,
|
||||
):
|
||||
"""
|
||||
Opens up a browser with advanced stealth features and fetch content using StealthyFetcher.
|
||||
|
||||
:param url: Target url.
|
||||
:param output_file: Output file path (.md for Markdown, .html for HTML).
|
||||
:param headless: Run the browser in headless/hidden, virtual screen mode, or headful/visible mode.
|
||||
:param block_images: Prevent the loading of images through Firefox preferences.
|
||||
:param disable_resources: Drop requests of unnecessary resources for a speed boost.
|
||||
:param block_webrtc: Blocks WebRTC entirely.
|
||||
:param humanize: Humanize the cursor movement.
|
||||
:param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page.
|
||||
:param allow_webgl: Allow WebGL (recommended to keep enabled).
|
||||
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
||||
:param disable_ads: Install the uBlock Origin addon on the browser.
|
||||
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page.
|
||||
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before returning.
|
||||
:param css_selector: CSS selector to extract specific content.
|
||||
:param wait_selector: Wait for a specific CSS selector to be in a specific state.
|
||||
:param geoip: Automatically use IP's longitude, latitude, timezone, country, locale.
|
||||
:param proxy: The proxy to be used with requests.
|
||||
:param extra_headers: Extra headers to add to the request.
|
||||
"""
|
||||
|
||||
# Parse parameters
|
||||
parsed_headers = parse_headers(extra_headers)
|
||||
|
||||
# Build request arguments
|
||||
kwargs = {
|
||||
"headless": headless,
|
||||
"block_images": block_images,
|
||||
"disable_resources": disable_resources,
|
||||
"block_webrtc": block_webrtc,
|
||||
"humanize": humanize,
|
||||
"solve_cloudflare": solve_cloudflare,
|
||||
"allow_webgl": allow_webgl,
|
||||
"network_idle": network_idle,
|
||||
"disable_ads": disable_ads,
|
||||
"timeout": timeout,
|
||||
"geoip": geoip,
|
||||
}
|
||||
|
||||
if wait > 0:
|
||||
kwargs["wait"] = wait
|
||||
if wait_selector:
|
||||
kwargs["wait_selector"] = wait_selector
|
||||
if proxy:
|
||||
kwargs["proxy"] = proxy
|
||||
if parsed_headers:
|
||||
kwargs["extra_headers"] = parsed_headers
|
||||
|
||||
make_request_and_save(
|
||||
StealthyFetcher.fetch, url, output_file, css_selector, **kwargs
|
||||
)
|
||||
|
||||
|
||||
@group()
|
||||
def main():
|
||||
pass
|
||||
@@ -85,3 +860,4 @@ def main():
|
||||
# Adding commands
|
||||
main.add_command(install)
|
||||
main.add_command(shell)
|
||||
main.add_command(extract)
|
||||
|
||||
+61
-11
@@ -1,4 +1,5 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
from re import sub as re_sub
|
||||
from sys import stderr
|
||||
from functools import wraps
|
||||
from http import cookies as Cookie
|
||||
@@ -24,6 +25,7 @@ from IPython.terminal.embed import InteractiveShellEmbed
|
||||
from orjson import loads as json_loads, JSONDecodeError
|
||||
|
||||
from scrapling import __version__
|
||||
from scrapling.core.custom_types import TextHandler
|
||||
from scrapling.core.utils import log
|
||||
from scrapling.parser import Adaptor, Adaptors
|
||||
from scrapling.core._types import List, Optional, Dict, Tuple, Any, Union
|
||||
@@ -63,6 +65,14 @@ Request = namedtuple(
|
||||
)
|
||||
|
||||
|
||||
def _CookieParser(cookie_string):
|
||||
# Errors will be handled on call so the log can be specified
|
||||
cookie_parser = Cookie.SimpleCookie()
|
||||
cookie_parser.load(cookie_string)
|
||||
for key, morsel in cookie_parser.items():
|
||||
yield key, morsel.value
|
||||
|
||||
|
||||
# Suppress exit on error to handle parsing errors gracefully
|
||||
class NoExitArgumentParser(ArgumentParser):
|
||||
def error(self, message):
|
||||
@@ -156,12 +166,11 @@ class CurlParser:
|
||||
|
||||
if header_key.lower() == "cookie":
|
||||
try:
|
||||
cookie_parser = Cookie.SimpleCookie()
|
||||
cookie_parser.load(header_value)
|
||||
for key, morsel in cookie_parser.items():
|
||||
cookie_dict[key] = morsel.value
|
||||
cookie_dict = {
|
||||
key: value for key, value in _CookieParser(header_value)
|
||||
}
|
||||
except Exception as e:
|
||||
log.error(
|
||||
raise ValueError(
|
||||
f"Could not parse cookie string from -H '{header_value}': {e}"
|
||||
)
|
||||
else:
|
||||
@@ -221,12 +230,9 @@ class CurlParser:
|
||||
if parsed_args.cookie:
|
||||
# We are focusing on the string format from DevTools.
|
||||
try:
|
||||
cookie_parser = Cookie.SimpleCookie()
|
||||
cookie_parser.load(parsed_args.cookie)
|
||||
for key, morsel in cookie_parser.items():
|
||||
# Update the cookie dict, potentially overwriting
|
||||
# cookies with the same name from -H 'Cookie:'
|
||||
cookies[key] = morsel.value
|
||||
for key, value in _CookieParser(parsed_args.cookie):
|
||||
# Update the cookie dict, potentially overwriting cookies with the same name from -H 'cookie:'
|
||||
cookies[key] = value
|
||||
log.debug(f"Parsed cookies from -b argument: {list(cookies.keys())}")
|
||||
except Exception as e:
|
||||
log.error(
|
||||
@@ -545,3 +551,47 @@ Type 'exit' or press Ctrl+D to exit.
|
||||
return
|
||||
|
||||
ipython_shell()
|
||||
|
||||
|
||||
class Convertor:
|
||||
"""Utils for the extract shell command"""
|
||||
|
||||
@classmethod
|
||||
def __convert_to_markdown(cls, body: TextHandler) -> str:
|
||||
"""Convert HTML content to Markdown"""
|
||||
from markdownify import markdownify
|
||||
|
||||
return markdownify(body)
|
||||
|
||||
@classmethod
|
||||
def write_content_to_file(
|
||||
cls, page: Adaptor, filename: str, css_selector: Optional[str] = None
|
||||
) -> None:
|
||||
"""Write an Adaptor's content to a file"""
|
||||
if not page or not isinstance(page, Adaptor):
|
||||
raise TypeError("Input must be of type `Adaptor`")
|
||||
elif not filename or not isinstance(filename, str) or not filename.strip():
|
||||
raise ValueError("Filename must be provided")
|
||||
elif not filename.endswith((".md", ".html", ".txt")):
|
||||
raise ValueError(
|
||||
"Unknown file type: filename must end with '.md', '.html', or '.txt'"
|
||||
)
|
||||
else:
|
||||
body = page if not css_selector else page.css_first(css_selector)
|
||||
with open(filename, "w", encoding="utf-8") as f:
|
||||
if filename.endswith(".md"):
|
||||
f.write(cls.__convert_to_markdown(body.body))
|
||||
elif filename.endswith(".html"):
|
||||
f.write(body.body)
|
||||
elif filename.endswith(".txt"):
|
||||
txt_content = body.get_all_text(strip=True)
|
||||
for s in (
|
||||
"\n",
|
||||
"\r",
|
||||
"\t",
|
||||
" ",
|
||||
):
|
||||
# Remove consecutive white-spaces
|
||||
txt_content = re_sub(f"[{s}]+", s, txt_content)
|
||||
|
||||
f.write(txt_content)
|
||||
|
||||
Reference in New Issue
Block a user