feat(extract): Adding new command to CLI options + Optimizations
Users can now fetch websites directly without code and extract full/selected HTML content as HTML, Markdown, or extract text content.
This commit is contained in:
+785
-9
@@ -1,21 +1,75 @@
|
|||||||
import os
|
|
||||||
import sys
|
|
||||||
import subprocess
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from subprocess import check_output
|
||||||
|
from sys import executable as python_executable
|
||||||
|
|
||||||
from click import command, option, Choice, group
|
from scrapling.core.utils import log
|
||||||
|
from scrapling.core.shell import Convertor, _CookieParser
|
||||||
|
from scrapling.fetchers import Fetcher, DynamicFetcher, StealthyFetcher
|
||||||
|
|
||||||
|
from orjson import loads as json_loads, JSONDecodeError
|
||||||
|
from click import command, option, Choice, group, argument
|
||||||
|
|
||||||
|
__OUTPUT_FILE_HELP__ = "Output file path can be HTML content, Markdown of the HTML content, or the text content. Use file extensions (`.html`/`.md`/`.txt`) respectively."
|
||||||
|
|
||||||
|
|
||||||
def get_package_dir():
|
def get_package_dir():
|
||||||
return Path(os.path.dirname(__file__))
|
return Path(__file__).parent
|
||||||
|
|
||||||
|
|
||||||
def run_command(cmd, line):
|
def run_command(cmd, line):
|
||||||
print(f"Installing {line}...")
|
print(f"Installing {line}...")
|
||||||
_ = subprocess.check_call(cmd, shell=False) # nosec B603
|
_ = check_output(cmd, shell=False) # nosec B603
|
||||||
# I meant to not use try except here
|
# I meant to not use try except here
|
||||||
|
|
||||||
|
|
||||||
|
def parse_headers(header_strings):
|
||||||
|
"""Parse header strings into a dictionary"""
|
||||||
|
headers = {}
|
||||||
|
for header in header_strings:
|
||||||
|
if ":" in header:
|
||||||
|
key, value = header.split(":", 1)
|
||||||
|
headers[key.strip()] = value.strip()
|
||||||
|
else:
|
||||||
|
log.warning(f"Invalid header format '{header}', should be 'Key: Value'")
|
||||||
|
return headers
|
||||||
|
|
||||||
|
|
||||||
|
def parse_cookies(cookie_string):
|
||||||
|
"""Parse cookie string into a dictionary"""
|
||||||
|
if not cookie_string:
|
||||||
|
return {}
|
||||||
|
|
||||||
|
try:
|
||||||
|
cookies = {key: value for key, value in _CookieParser(cookie_string)}
|
||||||
|
except Exception as e:
|
||||||
|
raise ValueError(f"Could not parse cookies '{cookie_string}': {e}")
|
||||||
|
|
||||||
|
return cookies
|
||||||
|
|
||||||
|
|
||||||
|
def parse_json_data(json_string):
|
||||||
|
"""Parse JSON string into a Python object"""
|
||||||
|
if not json_string:
|
||||||
|
return None
|
||||||
|
|
||||||
|
try:
|
||||||
|
return json_loads(json_string)
|
||||||
|
except JSONDecodeError as e:
|
||||||
|
raise ValueError(f"Invalid JSON data '{json_string}': {e}")
|
||||||
|
|
||||||
|
|
||||||
|
def make_request_and_save(fetcher_func, url, output_file, css_selector=None, **kwargs):
|
||||||
|
"""Make a request using the specified fetcher function and save the result"""
|
||||||
|
# Handle relative paths - convert to an absolute path based on the current working directory
|
||||||
|
output_path = Path(output_file)
|
||||||
|
if not output_path.is_absolute():
|
||||||
|
output_path = Path.cwd() / output_file
|
||||||
|
|
||||||
|
response = fetcher_func(url, **kwargs)
|
||||||
|
Convertor.write_content_to_file(response, str(output_path), css_selector)
|
||||||
|
log.info(f"Content successfully saved to '{output_path}'")
|
||||||
|
|
||||||
|
|
||||||
@command(help="Install all Scrapling's Fetchers dependencies")
|
@command(help="Install all Scrapling's Fetchers dependencies")
|
||||||
@option(
|
@option(
|
||||||
"-f",
|
"-f",
|
||||||
@@ -32,15 +86,22 @@ def install(force):
|
|||||||
or not get_package_dir().joinpath(".scrapling_dependencies_installed").exists()
|
or not get_package_dir().joinpath(".scrapling_dependencies_installed").exists()
|
||||||
):
|
):
|
||||||
run_command(
|
run_command(
|
||||||
[sys.executable, "-m", "playwright", "install", "chromium"],
|
[python_executable, "-m", "playwright", "install", "chromium"],
|
||||||
"Playwright browsers",
|
"Playwright browsers",
|
||||||
)
|
)
|
||||||
run_command(
|
run_command(
|
||||||
[sys.executable, "-m", "playwright", "install-deps", "chromium", "firefox"],
|
[
|
||||||
|
python_executable,
|
||||||
|
"-m",
|
||||||
|
"playwright",
|
||||||
|
"install-deps",
|
||||||
|
"chromium",
|
||||||
|
"firefox",
|
||||||
|
],
|
||||||
"Playwright dependencies",
|
"Playwright dependencies",
|
||||||
)
|
)
|
||||||
run_command(
|
run_command(
|
||||||
[sys.executable, "-m", "camoufox", "fetch", "--browserforge"],
|
[python_executable, "-m", "camoufox", "fetch", "--browserforge"],
|
||||||
"Camoufox browser and databases",
|
"Camoufox browser and databases",
|
||||||
)
|
)
|
||||||
# if no errors raised by the above commands, then we add the below file
|
# if no errors raised by the above commands, then we add the below file
|
||||||
@@ -77,6 +138,720 @@ def shell(code, level):
|
|||||||
console.start()
|
console.start()
|
||||||
|
|
||||||
|
|
||||||
|
def parse_extract_arguments(headers, cookies, params, json=None):
|
||||||
|
"""Parse arguments for extract command"""
|
||||||
|
parsed_headers = parse_headers(headers)
|
||||||
|
parsed_cookies = parse_cookies(cookies)
|
||||||
|
parsed_json = parse_json_data(json)
|
||||||
|
parsed_params = {}
|
||||||
|
for param in params:
|
||||||
|
if "=" in param:
|
||||||
|
key, value = param.split("=", 1)
|
||||||
|
parsed_params[key] = value
|
||||||
|
|
||||||
|
return parsed_headers, parsed_cookies, parsed_params, parsed_json
|
||||||
|
|
||||||
|
|
||||||
|
@group(
|
||||||
|
help="Fetch web pages using various fetchers and extract full/selected HTML content as HTML, Markdown, or extract text content."
|
||||||
|
)
|
||||||
|
def extract():
|
||||||
|
"""Extract content from web pages and save to files"""
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
@extract.command(
|
||||||
|
help=f"Perform a GET request and save content to file.\n\n{__OUTPUT_FILE_HELP__}"
|
||||||
|
)
|
||||||
|
@argument("url", required=True)
|
||||||
|
@argument("output_file", required=True)
|
||||||
|
@option(
|
||||||
|
"--headers",
|
||||||
|
"-H",
|
||||||
|
multiple=True,
|
||||||
|
help='HTTP headers in format "Key: Value" (can be used multiple times)',
|
||||||
|
)
|
||||||
|
@option("--cookies", help='Cookies string in format "name1=value1; name2=value2"')
|
||||||
|
@option(
|
||||||
|
"--timeout", type=int, default=30, help="Request timeout in seconds (default: 30)"
|
||||||
|
)
|
||||||
|
@option("--proxy", help='Proxy URL in format "http://username:password@host:port"')
|
||||||
|
@option(
|
||||||
|
"--css-selector",
|
||||||
|
"-s",
|
||||||
|
help="CSS selector to extract specific content from the page. It resolves to the first match if multiple matches are found.",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--params",
|
||||||
|
"-p",
|
||||||
|
multiple=True,
|
||||||
|
help='Query parameters in format "key=value" (can be used multiple times)',
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--follow-redirects/--no-follow-redirects",
|
||||||
|
default=True,
|
||||||
|
help="Whether to follow redirects (default: True)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--verify/--no-verify",
|
||||||
|
default=True,
|
||||||
|
help="Whether to verify SSL certificates (default: True)",
|
||||||
|
)
|
||||||
|
@option("--impersonate", help="Browser to impersonate (e.g., chrome, firefox).")
|
||||||
|
@option(
|
||||||
|
"--stealthy-headers/--no-stealthy-headers",
|
||||||
|
default=True,
|
||||||
|
help="Use stealthy browser headers (default: True)",
|
||||||
|
)
|
||||||
|
def get(
|
||||||
|
url,
|
||||||
|
output_file,
|
||||||
|
headers,
|
||||||
|
cookies,
|
||||||
|
timeout,
|
||||||
|
proxy,
|
||||||
|
css_selector,
|
||||||
|
params,
|
||||||
|
follow_redirects,
|
||||||
|
verify,
|
||||||
|
impersonate,
|
||||||
|
stealthy_headers,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Perform a GET request and save content to file.
|
||||||
|
|
||||||
|
:param url: Target URL for the request.
|
||||||
|
:param output_file: Output file path (.md for Markdown, .html for HTML).
|
||||||
|
:param headers: HTTP headers to include in the request.
|
||||||
|
:param cookies: Cookies to use in the request.
|
||||||
|
:param timeout: Number of seconds to wait before timing out.
|
||||||
|
:param proxy: Proxy URL to use. (Format: "http://username:password@localhost:8030")
|
||||||
|
:param css_selector: CSS selector to extract specific content.
|
||||||
|
:param params: Query string parameters for the request.
|
||||||
|
:param follow_redirects: Whether to follow redirects.
|
||||||
|
:param verify: Whether to verify HTTPS certificates.
|
||||||
|
:param impersonate: Browser version to impersonate.
|
||||||
|
:param stealthy_headers: If enabled, creates and adds real browser headers.
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Parse parameters
|
||||||
|
parsed_headers, parsed_cookies, parsed_params, _ = parse_extract_arguments(
|
||||||
|
headers, cookies, params
|
||||||
|
)
|
||||||
|
|
||||||
|
# Build request arguments
|
||||||
|
kwargs = {
|
||||||
|
"headers": parsed_headers if parsed_headers else None,
|
||||||
|
"cookies": parsed_cookies if parsed_cookies else None,
|
||||||
|
"timeout": timeout,
|
||||||
|
"follow_redirects": follow_redirects,
|
||||||
|
"verify": verify,
|
||||||
|
"stealthy_headers": stealthy_headers,
|
||||||
|
"impersonate": impersonate,
|
||||||
|
}
|
||||||
|
|
||||||
|
if parsed_params:
|
||||||
|
kwargs["params"] = parsed_params
|
||||||
|
if proxy:
|
||||||
|
kwargs["proxy"] = proxy
|
||||||
|
|
||||||
|
make_request_and_save(Fetcher.get, url, output_file, css_selector, **kwargs)
|
||||||
|
|
||||||
|
|
||||||
|
@extract.command(
|
||||||
|
help=f"Perform a POST request and save content to file.\n\n{__OUTPUT_FILE_HELP__}"
|
||||||
|
)
|
||||||
|
@argument("url", required=True)
|
||||||
|
@argument("output_file", required=True)
|
||||||
|
@option(
|
||||||
|
"--data",
|
||||||
|
"-d",
|
||||||
|
help='Form data to include in the request body (as string, ex: "param1=value1¶m2=value2")',
|
||||||
|
)
|
||||||
|
@option("--json", "-j", help="JSON data to include in the request body (as string)")
|
||||||
|
@option(
|
||||||
|
"--headers",
|
||||||
|
"-H",
|
||||||
|
multiple=True,
|
||||||
|
help='HTTP headers in format "Key: Value" (can be used multiple times)',
|
||||||
|
)
|
||||||
|
@option("--cookies", help='Cookies string in format "name1=value1; name2=value2"')
|
||||||
|
@option(
|
||||||
|
"--timeout", type=int, default=30, help="Request timeout in seconds (default: 30)"
|
||||||
|
)
|
||||||
|
@option("--proxy", help='Proxy URL in format "http://username:password@host:port"')
|
||||||
|
@option(
|
||||||
|
"--css-selector",
|
||||||
|
"-s",
|
||||||
|
help="CSS selector to extract specific content from the page",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--params",
|
||||||
|
"-p",
|
||||||
|
multiple=True,
|
||||||
|
help='Query parameters in format "key=value" (can be used multiple times)',
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--follow-redirects/--no-follow-redirects",
|
||||||
|
default=True,
|
||||||
|
help="Whether to follow redirects (default: True)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--verify/--no-verify",
|
||||||
|
default=True,
|
||||||
|
help="Whether to verify SSL certificates (default: True)",
|
||||||
|
)
|
||||||
|
@option("--impersonate", help="Browser to impersonate (e.g., chrome, firefox).")
|
||||||
|
@option(
|
||||||
|
"--stealthy-headers/--no-stealthy-headers",
|
||||||
|
default=True,
|
||||||
|
help="Use stealthy browser headers (default: True)",
|
||||||
|
)
|
||||||
|
def post(
|
||||||
|
url,
|
||||||
|
output_file,
|
||||||
|
data,
|
||||||
|
json,
|
||||||
|
headers,
|
||||||
|
cookies,
|
||||||
|
timeout,
|
||||||
|
proxy,
|
||||||
|
css_selector,
|
||||||
|
params,
|
||||||
|
follow_redirects,
|
||||||
|
verify,
|
||||||
|
impersonate,
|
||||||
|
stealthy_headers,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Perform a POST request and save content to file.
|
||||||
|
|
||||||
|
:param url: Target URL for the request.
|
||||||
|
:param output_file: Output file path (.md for Markdown, .html for HTML).
|
||||||
|
:param data: Form data to include in the request body. (as string, ex: "param1=value1¶m2=value2")
|
||||||
|
:param json: A JSON serializable object to include in the body of the request.
|
||||||
|
:param headers: Headers to include in the request.
|
||||||
|
:param cookies: Cookies to use in the request.
|
||||||
|
:param timeout: Number of seconds to wait before timing out.
|
||||||
|
:param proxy: Proxy URL to use.
|
||||||
|
:param css_selector: CSS selector to extract specific content.
|
||||||
|
:param params: Query string parameters for the request.
|
||||||
|
:param follow_redirects: Whether to follow redirects.
|
||||||
|
:param verify: Whether to verify HTTPS certificates.
|
||||||
|
:param impersonate: Browser version to impersonate.
|
||||||
|
:param stealthy_headers: If enabled, creates and adds real browser headers.
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Parse parameters
|
||||||
|
parsed_headers, parsed_cookies, parsed_params, parsed_json = (
|
||||||
|
parse_extract_arguments(headers, cookies, params, json)
|
||||||
|
)
|
||||||
|
|
||||||
|
# Build request arguments
|
||||||
|
kwargs = {
|
||||||
|
"headers": parsed_headers if parsed_headers else None,
|
||||||
|
"cookies": parsed_cookies if parsed_cookies else None,
|
||||||
|
"timeout": timeout,
|
||||||
|
"follow_redirects": follow_redirects,
|
||||||
|
"verify": verify,
|
||||||
|
"stealthy_headers": stealthy_headers,
|
||||||
|
"impersonate": impersonate,
|
||||||
|
}
|
||||||
|
|
||||||
|
if data:
|
||||||
|
kwargs["data"] = data
|
||||||
|
if parsed_json:
|
||||||
|
kwargs["json"] = parsed_json
|
||||||
|
if parsed_params:
|
||||||
|
kwargs["params"] = parsed_params
|
||||||
|
if proxy:
|
||||||
|
kwargs["proxy"] = proxy
|
||||||
|
|
||||||
|
make_request_and_save(Fetcher.post, url, output_file, css_selector, **kwargs)
|
||||||
|
|
||||||
|
|
||||||
|
@extract.command(
|
||||||
|
help=f"Perform a PUT request and save content to file.\n\n{__OUTPUT_FILE_HELP__}"
|
||||||
|
)
|
||||||
|
@argument("url", required=True)
|
||||||
|
@argument("output_file", required=True)
|
||||||
|
@option("--data", "-d", help="Form data to include in the request body")
|
||||||
|
@option("--json", "-j", help="JSON data to include in the request body (as string)")
|
||||||
|
@option(
|
||||||
|
"--headers",
|
||||||
|
"-H",
|
||||||
|
multiple=True,
|
||||||
|
help='HTTP headers in format "Key: Value" (can be used multiple times)',
|
||||||
|
)
|
||||||
|
@option("--cookies", help='Cookies string in format "name1=value1; name2=value2"')
|
||||||
|
@option(
|
||||||
|
"--timeout", type=int, default=30, help="Request timeout in seconds (default: 30)"
|
||||||
|
)
|
||||||
|
@option("--proxy", help='Proxy URL in format "http://username:password@host:port"')
|
||||||
|
@option(
|
||||||
|
"--css-selector",
|
||||||
|
"-s",
|
||||||
|
help="CSS selector to extract specific content from the page",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--params",
|
||||||
|
"-p",
|
||||||
|
multiple=True,
|
||||||
|
help='Query parameters in format "key=value" (can be used multiple times)',
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--follow-redirects/--no-follow-redirects",
|
||||||
|
default=True,
|
||||||
|
help="Whether to follow redirects (default: True)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--verify/--no-verify",
|
||||||
|
default=True,
|
||||||
|
help="Whether to verify SSL certificates (default: True)",
|
||||||
|
)
|
||||||
|
@option("--impersonate", help="Browser to impersonate (e.g., chrome, firefox).")
|
||||||
|
@option(
|
||||||
|
"--stealthy-headers/--no-stealthy-headers",
|
||||||
|
default=True,
|
||||||
|
help="Use stealthy browser headers (default: True)",
|
||||||
|
)
|
||||||
|
def put(
|
||||||
|
url,
|
||||||
|
output_file,
|
||||||
|
data,
|
||||||
|
json,
|
||||||
|
headers,
|
||||||
|
cookies,
|
||||||
|
timeout,
|
||||||
|
proxy,
|
||||||
|
css_selector,
|
||||||
|
params,
|
||||||
|
follow_redirects,
|
||||||
|
verify,
|
||||||
|
impersonate,
|
||||||
|
stealthy_headers,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Perform a PUT request and save content to file.
|
||||||
|
|
||||||
|
:param url: Target URL for the request.
|
||||||
|
:param output_file: Output file path (.md for Markdown, .html for HTML).
|
||||||
|
:param data: Form data to include in the request body.
|
||||||
|
:param json: A JSON serializable object to include in the body of the request.
|
||||||
|
:param headers: Headers to include in the request.
|
||||||
|
:param cookies: Cookies to use in the request.
|
||||||
|
:param timeout: Number of seconds to wait before timing out.
|
||||||
|
:param proxy: Proxy URL to use.
|
||||||
|
:param css_selector: CSS selector to extract specific content.
|
||||||
|
:param params: Query string parameters for the request.
|
||||||
|
:param follow_redirects: Whether to follow redirects.
|
||||||
|
:param verify: Whether to verify HTTPS certificates.
|
||||||
|
:param impersonate: Browser version to impersonate.
|
||||||
|
:param stealthy_headers: If enabled, creates and adds real browser headers.
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Parse parameters
|
||||||
|
parsed_headers, parsed_cookies, parsed_params, parsed_json = (
|
||||||
|
parse_extract_arguments(headers, cookies, params, json)
|
||||||
|
)
|
||||||
|
|
||||||
|
# Build request arguments
|
||||||
|
kwargs = {
|
||||||
|
"headers": parsed_headers if parsed_headers else None,
|
||||||
|
"cookies": parsed_cookies if parsed_cookies else None,
|
||||||
|
"timeout": timeout,
|
||||||
|
"follow_redirects": follow_redirects,
|
||||||
|
"verify": verify,
|
||||||
|
"stealthy_headers": stealthy_headers,
|
||||||
|
"impersonate": impersonate,
|
||||||
|
}
|
||||||
|
|
||||||
|
if data:
|
||||||
|
kwargs["data"] = data
|
||||||
|
if parsed_json:
|
||||||
|
kwargs["json"] = parsed_json
|
||||||
|
if parsed_params:
|
||||||
|
kwargs["params"] = parsed_params
|
||||||
|
if proxy:
|
||||||
|
kwargs["proxy"] = proxy
|
||||||
|
|
||||||
|
make_request_and_save(Fetcher.put, url, output_file, css_selector, **kwargs)
|
||||||
|
|
||||||
|
|
||||||
|
@extract.command(
|
||||||
|
help=f"Perform a DELETE request and save content to file.\n\n{__OUTPUT_FILE_HELP__}"
|
||||||
|
)
|
||||||
|
@argument("url", required=True)
|
||||||
|
@argument("output_file", required=True)
|
||||||
|
@option(
|
||||||
|
"--headers",
|
||||||
|
"-H",
|
||||||
|
multiple=True,
|
||||||
|
help='HTTP headers in format "Key: Value" (can be used multiple times)',
|
||||||
|
)
|
||||||
|
@option("--cookies", help='Cookies string in format "name1=value1; name2=value2"')
|
||||||
|
@option(
|
||||||
|
"--timeout", type=int, default=30, help="Request timeout in seconds (default: 30)"
|
||||||
|
)
|
||||||
|
@option("--proxy", help='Proxy URL in format "http://username:password@host:port"')
|
||||||
|
@option(
|
||||||
|
"--css-selector",
|
||||||
|
"-s",
|
||||||
|
help="CSS selector to extract specific content from the page",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--params",
|
||||||
|
"-p",
|
||||||
|
multiple=True,
|
||||||
|
help='Query parameters in format "key=value" (can be used multiple times)',
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--follow-redirects/--no-follow-redirects",
|
||||||
|
default=True,
|
||||||
|
help="Whether to follow redirects (default: True)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--verify/--no-verify",
|
||||||
|
default=True,
|
||||||
|
help="Whether to verify SSL certificates (default: True)",
|
||||||
|
)
|
||||||
|
@option("--impersonate", help="Browser to impersonate (e.g., chrome, firefox).")
|
||||||
|
@option(
|
||||||
|
"--stealthy-headers/--no-stealthy-headers",
|
||||||
|
default=True,
|
||||||
|
help="Use stealthy browser headers (default: True)",
|
||||||
|
)
|
||||||
|
def delete(
|
||||||
|
url,
|
||||||
|
output_file,
|
||||||
|
headers,
|
||||||
|
cookies,
|
||||||
|
timeout,
|
||||||
|
proxy,
|
||||||
|
css_selector,
|
||||||
|
params,
|
||||||
|
follow_redirects,
|
||||||
|
verify,
|
||||||
|
impersonate,
|
||||||
|
stealthy_headers,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Perform a DELETE request and save content to file.
|
||||||
|
|
||||||
|
:param url: Target URL for the request.
|
||||||
|
:param output_file: Output file path (.md for Markdown, .html for HTML).
|
||||||
|
:param headers: Headers to include in the request.
|
||||||
|
:param cookies: Cookies to use in the request.
|
||||||
|
:param timeout: Number of seconds to wait before timing out.
|
||||||
|
:param proxy: Proxy URL to use.
|
||||||
|
:param css_selector: CSS selector to extract specific content.
|
||||||
|
:param params: Query string parameters for the request.
|
||||||
|
:param follow_redirects: Whether to follow redirects.
|
||||||
|
:param verify: Whether to verify HTTPS certificates.
|
||||||
|
:param impersonate: Browser version to impersonate.
|
||||||
|
:param stealthy_headers: If enabled, creates and adds real browser headers.
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Parse parameters
|
||||||
|
parsed_headers, parsed_cookies, parsed_params, _ = parse_extract_arguments(
|
||||||
|
headers, cookies, params
|
||||||
|
)
|
||||||
|
|
||||||
|
# Build request arguments
|
||||||
|
kwargs = {
|
||||||
|
"headers": parsed_headers if parsed_headers else None,
|
||||||
|
"cookies": parsed_cookies if parsed_cookies else None,
|
||||||
|
"timeout": timeout,
|
||||||
|
"follow_redirects": follow_redirects,
|
||||||
|
"verify": verify,
|
||||||
|
"stealthy_headers": stealthy_headers,
|
||||||
|
"impersonate": impersonate,
|
||||||
|
}
|
||||||
|
|
||||||
|
if parsed_params:
|
||||||
|
kwargs["params"] = parsed_params
|
||||||
|
if proxy:
|
||||||
|
kwargs["proxy"] = proxy
|
||||||
|
|
||||||
|
make_request_and_save(Fetcher.delete, url, output_file, css_selector, **kwargs)
|
||||||
|
|
||||||
|
|
||||||
|
@extract.command(
|
||||||
|
help=f"Use DynamicFetcher to fetch content with browser automation.\n\n{__OUTPUT_FILE_HELP__}"
|
||||||
|
)
|
||||||
|
@argument("url", required=True)
|
||||||
|
@argument("output_file", required=True)
|
||||||
|
@option(
|
||||||
|
"--headless/--no-headless",
|
||||||
|
default=True,
|
||||||
|
help="Run browser in headless mode (default: True)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--disable-resources/--enable-resources",
|
||||||
|
default=False,
|
||||||
|
help="Drop unnecessary resources for speed boost (default: False)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--network-idle/--no-network-idle",
|
||||||
|
default=False,
|
||||||
|
help="Wait for network idle (default: False)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--timeout",
|
||||||
|
type=int,
|
||||||
|
default=30000,
|
||||||
|
help="Timeout in milliseconds (default: 30000)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--wait",
|
||||||
|
type=int,
|
||||||
|
default=0,
|
||||||
|
help="Additional wait time in milliseconds after page load (default: 0)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--css-selector",
|
||||||
|
"-s",
|
||||||
|
help="CSS selector to extract specific content from the page",
|
||||||
|
)
|
||||||
|
@option("--wait-selector", help="CSS selector to wait for before proceeding")
|
||||||
|
@option("--locale", default="en-US", help="Browser locale (default: en-US)")
|
||||||
|
@option(
|
||||||
|
"--stealth/--no-stealth", default=False, help="Enable stealth mode (default: False)"
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--hide-canvas/--show-canvas",
|
||||||
|
default=False,
|
||||||
|
help="Add noise to canvas operations (default: False)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--disable-webgl/--enable-webgl",
|
||||||
|
default=False,
|
||||||
|
help="Disable WebGL support (default: False)",
|
||||||
|
)
|
||||||
|
@option("--proxy", help='Proxy URL in format "http://username:password@host:port"')
|
||||||
|
@option(
|
||||||
|
"--extra-headers",
|
||||||
|
"-H",
|
||||||
|
multiple=True,
|
||||||
|
help='Extra headers in format "Key: Value" (can be used multiple times)',
|
||||||
|
)
|
||||||
|
def fetch(
|
||||||
|
url,
|
||||||
|
output_file,
|
||||||
|
headless,
|
||||||
|
disable_resources,
|
||||||
|
network_idle,
|
||||||
|
timeout,
|
||||||
|
wait,
|
||||||
|
css_selector,
|
||||||
|
wait_selector,
|
||||||
|
locale,
|
||||||
|
stealth,
|
||||||
|
hide_canvas,
|
||||||
|
disable_webgl,
|
||||||
|
proxy,
|
||||||
|
extra_headers,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Opens up a browser and fetch content using DynamicFetcher.
|
||||||
|
|
||||||
|
:param url: Target url.
|
||||||
|
:param output_file: Output file path (.md for Markdown, .html for HTML).
|
||||||
|
:param headless: Run the browser in headless/hidden or headful/visible mode.
|
||||||
|
:param disable_resources: Drop requests of unnecessary resources for a speed boost.
|
||||||
|
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
||||||
|
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page.
|
||||||
|
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before returning.
|
||||||
|
:param css_selector: CSS selector to extract specific content.
|
||||||
|
:param wait_selector: Wait for a specific CSS selector to be in a specific state.
|
||||||
|
:param locale: Set the locale for the browser.
|
||||||
|
:param stealth: Enables stealth mode.
|
||||||
|
:param hide_canvas: Add random noise to canvas operations to prevent fingerprinting.
|
||||||
|
:param disable_webgl: Disables WebGL and WebGL 2.0 support entirely.
|
||||||
|
:param proxy: The proxy to be used with requests.
|
||||||
|
:param extra_headers: Extra headers to add to the request.
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Parse parameters
|
||||||
|
parsed_headers = parse_headers(extra_headers)
|
||||||
|
|
||||||
|
# Build request arguments
|
||||||
|
kwargs = {
|
||||||
|
"headless": headless,
|
||||||
|
"disable_resources": disable_resources,
|
||||||
|
"network_idle": network_idle,
|
||||||
|
"timeout": timeout,
|
||||||
|
"locale": locale,
|
||||||
|
"stealth": stealth,
|
||||||
|
"hide_canvas": hide_canvas,
|
||||||
|
"disable_webgl": disable_webgl,
|
||||||
|
}
|
||||||
|
|
||||||
|
if wait > 0:
|
||||||
|
kwargs["wait"] = wait
|
||||||
|
if wait_selector:
|
||||||
|
kwargs["wait_selector"] = wait_selector
|
||||||
|
if proxy:
|
||||||
|
kwargs["proxy"] = proxy
|
||||||
|
if parsed_headers:
|
||||||
|
kwargs["extra_headers"] = parsed_headers
|
||||||
|
|
||||||
|
make_request_and_save(
|
||||||
|
DynamicFetcher.fetch, url, output_file, css_selector, **kwargs
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@extract.command(
|
||||||
|
help=f"Use StealthyFetcher to fetch content with advanced stealth features.\n\n{__OUTPUT_FILE_HELP__}"
|
||||||
|
)
|
||||||
|
@argument("url", required=True)
|
||||||
|
@argument("output_file", required=True)
|
||||||
|
@option(
|
||||||
|
"--headless/--no-headless",
|
||||||
|
default=True,
|
||||||
|
help="Run browser in headless mode (default: True)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--block-images/--allow-images",
|
||||||
|
default=False,
|
||||||
|
help="Block image loading (default: False)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--disable-resources/--enable-resources",
|
||||||
|
default=False,
|
||||||
|
help="Drop unnecessary resources for speed boost (default: False)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--block-webrtc/--allow-webrtc",
|
||||||
|
default=False,
|
||||||
|
help="Block WebRTC entirely (default: False)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--humanize/--no-humanize",
|
||||||
|
default=False,
|
||||||
|
help="Humanize cursor movement (default: False)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--solve-cloudflare/--no-solve-cloudflare",
|
||||||
|
default=False,
|
||||||
|
help="Solve Cloudflare challenges (default: False)",
|
||||||
|
)
|
||||||
|
@option("--allow-webgl/--block-webgl", default=True, help="Allow WebGL (default: True)")
|
||||||
|
@option(
|
||||||
|
"--network-idle/--no-network-idle",
|
||||||
|
default=False,
|
||||||
|
help="Wait for network idle (default: False)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--disable-ads/--allow-ads",
|
||||||
|
default=False,
|
||||||
|
help="Install uBlock Origin addon (default: False)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--timeout",
|
||||||
|
type=int,
|
||||||
|
default=30000,
|
||||||
|
help="Timeout in milliseconds (default: 30000)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--wait",
|
||||||
|
type=int,
|
||||||
|
default=0,
|
||||||
|
help="Additional wait time in milliseconds after page load (default: 0)",
|
||||||
|
)
|
||||||
|
@option(
|
||||||
|
"--css-selector",
|
||||||
|
"-s",
|
||||||
|
help="CSS selector to extract specific content from the page",
|
||||||
|
)
|
||||||
|
@option("--wait-selector", help="CSS selector to wait for before proceeding")
|
||||||
|
@option(
|
||||||
|
"--geoip/--no-geoip",
|
||||||
|
default=False,
|
||||||
|
help="Use IP geolocation for timezone/locale (default: False)",
|
||||||
|
)
|
||||||
|
@option("--proxy", help='Proxy URL in format "http://username:password@host:port"')
|
||||||
|
@option(
|
||||||
|
"--extra-headers",
|
||||||
|
"-H",
|
||||||
|
multiple=True,
|
||||||
|
help='Extra headers in format "Key: Value" (can be used multiple times)',
|
||||||
|
)
|
||||||
|
def stealthy_fetch(
|
||||||
|
url,
|
||||||
|
output_file,
|
||||||
|
headless,
|
||||||
|
block_images,
|
||||||
|
disable_resources,
|
||||||
|
block_webrtc,
|
||||||
|
humanize,
|
||||||
|
solve_cloudflare,
|
||||||
|
allow_webgl,
|
||||||
|
network_idle,
|
||||||
|
disable_ads,
|
||||||
|
timeout,
|
||||||
|
wait,
|
||||||
|
css_selector,
|
||||||
|
wait_selector,
|
||||||
|
geoip,
|
||||||
|
proxy,
|
||||||
|
extra_headers,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Opens up a browser with advanced stealth features and fetch content using StealthyFetcher.
|
||||||
|
|
||||||
|
:param url: Target url.
|
||||||
|
:param output_file: Output file path (.md for Markdown, .html for HTML).
|
||||||
|
:param headless: Run the browser in headless/hidden, virtual screen mode, or headful/visible mode.
|
||||||
|
:param block_images: Prevent the loading of images through Firefox preferences.
|
||||||
|
:param disable_resources: Drop requests of unnecessary resources for a speed boost.
|
||||||
|
:param block_webrtc: Blocks WebRTC entirely.
|
||||||
|
:param humanize: Humanize the cursor movement.
|
||||||
|
:param solve_cloudflare: Solves all 3 types of the Cloudflare's Turnstile wait page.
|
||||||
|
:param allow_webgl: Allow WebGL (recommended to keep enabled).
|
||||||
|
:param network_idle: Wait for the page until there are no network connections for at least 500 ms.
|
||||||
|
:param disable_ads: Install the uBlock Origin addon on the browser.
|
||||||
|
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page.
|
||||||
|
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before returning.
|
||||||
|
:param css_selector: CSS selector to extract specific content.
|
||||||
|
:param wait_selector: Wait for a specific CSS selector to be in a specific state.
|
||||||
|
:param geoip: Automatically use IP's longitude, latitude, timezone, country, locale.
|
||||||
|
:param proxy: The proxy to be used with requests.
|
||||||
|
:param extra_headers: Extra headers to add to the request.
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Parse parameters
|
||||||
|
parsed_headers = parse_headers(extra_headers)
|
||||||
|
|
||||||
|
# Build request arguments
|
||||||
|
kwargs = {
|
||||||
|
"headless": headless,
|
||||||
|
"block_images": block_images,
|
||||||
|
"disable_resources": disable_resources,
|
||||||
|
"block_webrtc": block_webrtc,
|
||||||
|
"humanize": humanize,
|
||||||
|
"solve_cloudflare": solve_cloudflare,
|
||||||
|
"allow_webgl": allow_webgl,
|
||||||
|
"network_idle": network_idle,
|
||||||
|
"disable_ads": disable_ads,
|
||||||
|
"timeout": timeout,
|
||||||
|
"geoip": geoip,
|
||||||
|
}
|
||||||
|
|
||||||
|
if wait > 0:
|
||||||
|
kwargs["wait"] = wait
|
||||||
|
if wait_selector:
|
||||||
|
kwargs["wait_selector"] = wait_selector
|
||||||
|
if proxy:
|
||||||
|
kwargs["proxy"] = proxy
|
||||||
|
if parsed_headers:
|
||||||
|
kwargs["extra_headers"] = parsed_headers
|
||||||
|
|
||||||
|
make_request_and_save(
|
||||||
|
StealthyFetcher.fetch, url, output_file, css_selector, **kwargs
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@group()
|
@group()
|
||||||
def main():
|
def main():
|
||||||
pass
|
pass
|
||||||
@@ -85,3 +860,4 @@ def main():
|
|||||||
# Adding commands
|
# Adding commands
|
||||||
main.add_command(install)
|
main.add_command(install)
|
||||||
main.add_command(shell)
|
main.add_command(shell)
|
||||||
|
main.add_command(extract)
|
||||||
|
|||||||
+61
-11
@@ -1,4 +1,5 @@
|
|||||||
# -*- coding: utf-8 -*-
|
# -*- coding: utf-8 -*-
|
||||||
|
from re import sub as re_sub
|
||||||
from sys import stderr
|
from sys import stderr
|
||||||
from functools import wraps
|
from functools import wraps
|
||||||
from http import cookies as Cookie
|
from http import cookies as Cookie
|
||||||
@@ -24,6 +25,7 @@ from IPython.terminal.embed import InteractiveShellEmbed
|
|||||||
from orjson import loads as json_loads, JSONDecodeError
|
from orjson import loads as json_loads, JSONDecodeError
|
||||||
|
|
||||||
from scrapling import __version__
|
from scrapling import __version__
|
||||||
|
from scrapling.core.custom_types import TextHandler
|
||||||
from scrapling.core.utils import log
|
from scrapling.core.utils import log
|
||||||
from scrapling.parser import Adaptor, Adaptors
|
from scrapling.parser import Adaptor, Adaptors
|
||||||
from scrapling.core._types import List, Optional, Dict, Tuple, Any, Union
|
from scrapling.core._types import List, Optional, Dict, Tuple, Any, Union
|
||||||
@@ -63,6 +65,14 @@ Request = namedtuple(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _CookieParser(cookie_string):
|
||||||
|
# Errors will be handled on call so the log can be specified
|
||||||
|
cookie_parser = Cookie.SimpleCookie()
|
||||||
|
cookie_parser.load(cookie_string)
|
||||||
|
for key, morsel in cookie_parser.items():
|
||||||
|
yield key, morsel.value
|
||||||
|
|
||||||
|
|
||||||
# Suppress exit on error to handle parsing errors gracefully
|
# Suppress exit on error to handle parsing errors gracefully
|
||||||
class NoExitArgumentParser(ArgumentParser):
|
class NoExitArgumentParser(ArgumentParser):
|
||||||
def error(self, message):
|
def error(self, message):
|
||||||
@@ -156,12 +166,11 @@ class CurlParser:
|
|||||||
|
|
||||||
if header_key.lower() == "cookie":
|
if header_key.lower() == "cookie":
|
||||||
try:
|
try:
|
||||||
cookie_parser = Cookie.SimpleCookie()
|
cookie_dict = {
|
||||||
cookie_parser.load(header_value)
|
key: value for key, value in _CookieParser(header_value)
|
||||||
for key, morsel in cookie_parser.items():
|
}
|
||||||
cookie_dict[key] = morsel.value
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
log.error(
|
raise ValueError(
|
||||||
f"Could not parse cookie string from -H '{header_value}': {e}"
|
f"Could not parse cookie string from -H '{header_value}': {e}"
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
@@ -221,12 +230,9 @@ class CurlParser:
|
|||||||
if parsed_args.cookie:
|
if parsed_args.cookie:
|
||||||
# We are focusing on the string format from DevTools.
|
# We are focusing on the string format from DevTools.
|
||||||
try:
|
try:
|
||||||
cookie_parser = Cookie.SimpleCookie()
|
for key, value in _CookieParser(parsed_args.cookie):
|
||||||
cookie_parser.load(parsed_args.cookie)
|
# Update the cookie dict, potentially overwriting cookies with the same name from -H 'cookie:'
|
||||||
for key, morsel in cookie_parser.items():
|
cookies[key] = value
|
||||||
# Update the cookie dict, potentially overwriting
|
|
||||||
# cookies with the same name from -H 'Cookie:'
|
|
||||||
cookies[key] = morsel.value
|
|
||||||
log.debug(f"Parsed cookies from -b argument: {list(cookies.keys())}")
|
log.debug(f"Parsed cookies from -b argument: {list(cookies.keys())}")
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
log.error(
|
log.error(
|
||||||
@@ -545,3 +551,47 @@ Type 'exit' or press Ctrl+D to exit.
|
|||||||
return
|
return
|
||||||
|
|
||||||
ipython_shell()
|
ipython_shell()
|
||||||
|
|
||||||
|
|
||||||
|
class Convertor:
|
||||||
|
"""Utils for the extract shell command"""
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def __convert_to_markdown(cls, body: TextHandler) -> str:
|
||||||
|
"""Convert HTML content to Markdown"""
|
||||||
|
from markdownify import markdownify
|
||||||
|
|
||||||
|
return markdownify(body)
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def write_content_to_file(
|
||||||
|
cls, page: Adaptor, filename: str, css_selector: Optional[str] = None
|
||||||
|
) -> None:
|
||||||
|
"""Write an Adaptor's content to a file"""
|
||||||
|
if not page or not isinstance(page, Adaptor):
|
||||||
|
raise TypeError("Input must be of type `Adaptor`")
|
||||||
|
elif not filename or not isinstance(filename, str) or not filename.strip():
|
||||||
|
raise ValueError("Filename must be provided")
|
||||||
|
elif not filename.endswith((".md", ".html", ".txt")):
|
||||||
|
raise ValueError(
|
||||||
|
"Unknown file type: filename must end with '.md', '.html', or '.txt'"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
body = page if not css_selector else page.css_first(css_selector)
|
||||||
|
with open(filename, "w", encoding="utf-8") as f:
|
||||||
|
if filename.endswith(".md"):
|
||||||
|
f.write(cls.__convert_to_markdown(body.body))
|
||||||
|
elif filename.endswith(".html"):
|
||||||
|
f.write(body.body)
|
||||||
|
elif filename.endswith(".txt"):
|
||||||
|
txt_content = body.get_all_text(strip=True)
|
||||||
|
for s in (
|
||||||
|
"\n",
|
||||||
|
"\r",
|
||||||
|
"\t",
|
||||||
|
" ",
|
||||||
|
):
|
||||||
|
# Remove consecutive white-spaces
|
||||||
|
txt_content = re_sub(f"[{s}]+", s, txt_content)
|
||||||
|
|
||||||
|
f.write(txt_content)
|
||||||
|
|||||||
Reference in New Issue
Block a user