style: type hints corrections and docstrings
This commit is contained in:
@@ -150,7 +150,7 @@ Tired of your PC slowing you down? Can’t keep your machine on 24/7 for scrapin
|
|||||||
```python
|
```python
|
||||||
from scrapling.fetchers import Fetcher
|
from scrapling.fetchers import Fetcher
|
||||||
|
|
||||||
# Do HTTP GET request to a web page and create an Selector instance
|
# Do HTTP GET request to a web page and create a Selector instance
|
||||||
page = Fetcher.get('https://quotes.toscrape.com/', stealthy_headers=True)
|
page = Fetcher.get('https://quotes.toscrape.com/', stealthy_headers=True)
|
||||||
# Get all text content from all HTML tags in the page except the `script` and `style` tags
|
# Get all text content from all HTML tags in the page except the `script` and `style` tags
|
||||||
page.get_all_text(ignore_tags=('script', 'style'))
|
page.get_all_text(ignore_tags=('script', 'style'))
|
||||||
|
|||||||
+3
-2
@@ -3,6 +3,7 @@ from subprocess import check_output
|
|||||||
from sys import executable as python_executable
|
from sys import executable as python_executable
|
||||||
|
|
||||||
from scrapling.core.utils import log
|
from scrapling.core.utils import log
|
||||||
|
from scrapling.engines.toolbelt import Response
|
||||||
from scrapling.core._types import List, Optional, Dict, Tuple, Any, Callable
|
from scrapling.core._types import List, Optional, Dict, Tuple, Any, Callable
|
||||||
from scrapling.fetchers import Fetcher, DynamicFetcher, StealthyFetcher
|
from scrapling.fetchers import Fetcher, DynamicFetcher, StealthyFetcher
|
||||||
from scrapling.core.shell import Convertor, _CookieParser, _ParseHeaders
|
from scrapling.core.shell import Convertor, _CookieParser, _ParseHeaders
|
||||||
@@ -32,12 +33,12 @@ def __ParseJSONData(json_string: Optional[str] = None) -> Optional[Dict[str, Any
|
|||||||
|
|
||||||
|
|
||||||
def __Request_and_Save(
|
def __Request_and_Save(
|
||||||
fetcher_func: Callable,
|
fetcher_func: Callable[..., Response],
|
||||||
url: str,
|
url: str,
|
||||||
output_file: str,
|
output_file: str,
|
||||||
css_selector: Optional[str] = None,
|
css_selector: Optional[str] = None,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
) -> None:
|
||||||
"""Make a request using the specified fetcher function and save the result"""
|
"""Make a request using the specified fetcher function and save the result"""
|
||||||
# Handle relative paths - convert to an absolute path based on the current working directory
|
# Handle relative paths - convert to an absolute path based on the current working directory
|
||||||
output_path = Path(output_file)
|
output_path = Path(output_file)
|
||||||
|
|||||||
@@ -4,6 +4,7 @@ Type definitions for type checking purposes.
|
|||||||
|
|
||||||
from typing import (
|
from typing import (
|
||||||
TYPE_CHECKING,
|
TYPE_CHECKING,
|
||||||
|
cast,
|
||||||
overload,
|
overload,
|
||||||
Any,
|
Any,
|
||||||
Callable,
|
Callable,
|
||||||
@@ -32,8 +33,13 @@ extraction_types = Literal["text", "html", "markdown"]
|
|||||||
StrOrBytes = Union[str, bytes]
|
StrOrBytes = Union[str, bytes]
|
||||||
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
try:
|
||||||
# typing.Self requires Python 3.11
|
# Python 3.11+
|
||||||
from typing_extensions import Self
|
from typing import Self # novermin
|
||||||
else:
|
except ImportError:
|
||||||
Self = object
|
try:
|
||||||
|
from typing_extensions import Self # Backport
|
||||||
|
except ImportError:
|
||||||
|
from typing import TypeVar
|
||||||
|
|
||||||
|
Self = object
|
||||||
|
|||||||
@@ -63,7 +63,7 @@ class ScraplingMCPServer:
|
|||||||
main_content_only: bool = True,
|
main_content_only: bool = True,
|
||||||
params: Optional[Union[Dict, List, Tuple]] = None,
|
params: Optional[Union[Dict, List, Tuple]] = None,
|
||||||
headers: Optional[Mapping[str, Optional[str]]] = None,
|
headers: Optional[Mapping[str, Optional[str]]] = None,
|
||||||
cookies: Optional[Union[dict[str, str], list[tuple[str, str]]]] = None,
|
cookies: Optional[Union[Dict[str, str], list[tuple[str, str]]]] = None,
|
||||||
timeout: Optional[Union[int, float]] = 30,
|
timeout: Optional[Union[int, float]] = 30,
|
||||||
follow_redirects: bool = True,
|
follow_redirects: bool = True,
|
||||||
max_redirects: int = 30,
|
max_redirects: int = 30,
|
||||||
@@ -142,7 +142,7 @@ class ScraplingMCPServer:
|
|||||||
main_content_only: bool = True,
|
main_content_only: bool = True,
|
||||||
params: Optional[Union[Dict, List, Tuple]] = None,
|
params: Optional[Union[Dict, List, Tuple]] = None,
|
||||||
headers: Optional[Mapping[str, Optional[str]]] = None,
|
headers: Optional[Mapping[str, Optional[str]]] = None,
|
||||||
cookies: Optional[Union[dict[str, str], list[tuple[str, str]]]] = None,
|
cookies: Optional[Union[Dict[str, str], list[tuple[str, str]]]] = None,
|
||||||
timeout: Optional[Union[int, float]] = 30,
|
timeout: Optional[Union[int, float]] = 30,
|
||||||
follow_redirects: bool = True,
|
follow_redirects: bool = True,
|
||||||
max_redirects: int = 30,
|
max_redirects: int = 30,
|
||||||
|
|||||||
@@ -1,9 +1,13 @@
|
|||||||
class SelectorsGeneration:
|
class SelectorsGeneration:
|
||||||
"""Selectors generation functions
|
"""
|
||||||
|
Functions for generating selectors
|
||||||
Trying to generate selectors like Firefox or maybe cleaner ones!? Ehm
|
Trying to generate selectors like Firefox or maybe cleaner ones!? Ehm
|
||||||
Inspiration: https://searchfox.org/mozilla-central/source/devtools/shared/inspector/css-logic.js#591"""
|
Inspiration: https://searchfox.org/mozilla-central/source/devtools/shared/inspector/css-logic.js#591
|
||||||
|
"""
|
||||||
|
|
||||||
def __general_selection(self, selection: str = "css", full_path=False) -> str:
|
def __general_selection(
|
||||||
|
self, selection: str = "css", full_path: bool = False
|
||||||
|
) -> str:
|
||||||
"""Generate a selector for the current element.
|
"""Generate a selector for the current element.
|
||||||
:return: A string of the generated selector.
|
:return: A string of the generated selector.
|
||||||
"""
|
"""
|
||||||
@@ -80,7 +84,7 @@ class SelectorsGeneration:
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def generate_xpath_selector(self) -> str:
|
def generate_xpath_selector(self) -> str:
|
||||||
"""Generate a XPath selector for the current element
|
"""Generate an XPath selector for the current element
|
||||||
:return: A string of the generated selector.
|
:return: A string of the generated selector.
|
||||||
"""
|
"""
|
||||||
return self.__general_selection("xpath")
|
return self.__general_selection("xpath")
|
||||||
|
|||||||
@@ -570,7 +570,7 @@ Type 'exit' or press Ctrl+D to exit.
|
|||||||
class Convertor:
|
class Convertor:
|
||||||
"""Utils for the extract shell command"""
|
"""Utils for the extract shell command"""
|
||||||
|
|
||||||
_extension_map: dict[str, extraction_types] = {
|
_extension_map: Dict[str, extraction_types] = {
|
||||||
"md": "markdown",
|
"md": "markdown",
|
||||||
"html": "html",
|
"html": "html",
|
||||||
"txt": "text",
|
"txt": "text",
|
||||||
@@ -591,7 +591,7 @@ class Convertor:
|
|||||||
css_selector: Optional[str] = None,
|
css_selector: Optional[str] = None,
|
||||||
main_content_only: bool = False,
|
main_content_only: bool = False,
|
||||||
) -> Generator[str, None, None]:
|
) -> Generator[str, None, None]:
|
||||||
"""Extract the content of an Selector"""
|
"""Extract the content of a Selector"""
|
||||||
if not page or not isinstance(page, Selector):
|
if not page or not isinstance(page, Selector):
|
||||||
raise TypeError("Input must be of type `Selector`")
|
raise TypeError("Input must be of type `Selector`")
|
||||||
elif not extraction_type or extraction_type not in cls._extension_map.values():
|
elif not extraction_type or extraction_type not in cls._extension_map.values():
|
||||||
@@ -624,7 +624,7 @@ class Convertor:
|
|||||||
def write_content_to_file(
|
def write_content_to_file(
|
||||||
cls, page: Selector, filename: str, css_selector: Optional[str] = None
|
cls, page: Selector, filename: str, css_selector: Optional[str] = None
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Write an Selector's content to a file"""
|
"""Write a Selector's content to a file"""
|
||||||
if not page or not isinstance(page, Selector):
|
if not page or not isinstance(page, Selector):
|
||||||
raise TypeError("Input must be of type `Selector`")
|
raise TypeError("Input must be of type `Selector`")
|
||||||
elif not filename or not isinstance(filename, str) or not filename.strip():
|
elif not filename or not isinstance(filename, str) or not filename.strip():
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
"""
|
"""
|
||||||
Most of this file is adapted version of the translator of parsel library with some modifications simply for 1 important reason...
|
Most of this file is an adapted version of the parsel library's translator with some modifications simply for 1 important reason...
|
||||||
|
|
||||||
To add pseudo-elements ``::text`` and ``::attr(ATTR_NAME)`` so we match Parsel/Scrapy selectors format which will be important in future releases but most importantly...
|
To add pseudo-elements ``::text`` and ``::attr(ATTR_NAME)`` so we match the Parsel/Scrapy selectors format which will be important in future releases but most importantly...
|
||||||
|
|
||||||
So you don't have to learn a new selectors/api method like what bs4 done with soupsieve :)
|
So you don't have to learn a new selectors/api method like what bs4 done with soupsieve :)
|
||||||
|
|
||||||
if you want to learn about this, head to https://cssselect.readthedocs.io/en/latest/#cssselect.FunctionalPseudoElement
|
If you want to learn about this, head to https://cssselect.readthedocs.io/en/latest/#cssselect.FunctionalPseudoElement
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import re
|
import re
|
||||||
|
|||||||
Reference in New Issue
Block a user