Sorting all library imports with isort

This commit is contained in:
Karim shoair
2024-11-30 14:02:06 +02:00
parent 26bfa88e50
commit 1a17b2ce46
24 changed files with 110 additions and 115 deletions
+9 -8
View File
@@ -1,17 +1,18 @@
import functools
import time import time
import timeit import timeit
import functools
import requests
from statistics import mean from statistics import mean
from scrapling import Adaptor import requests
from parsel import Selector
from lxml import etree, html
from bs4 import BeautifulSoup
from pyquery import PyQuery as pq
from autoscraper import AutoScraper from autoscraper import AutoScraper
from selectolax.parser import HTMLParser from bs4 import BeautifulSoup
from lxml import etree, html
from mechanicalsoup import StatefulBrowser from mechanicalsoup import StatefulBrowser
from parsel import Selector
from pyquery import PyQuery as pq
from selectolax.parser import HTMLParser
from scrapling import Adaptor
large_html = '<html><body>' + '<div class="item">' * 5000 + '</div>' * 5000 + '</body></html>' large_html = '<html><body>' + '<div class="item">' * 5000 + '</div>' * 5000 + '</body></html>'
+1 -1
View File
@@ -4,6 +4,7 @@ I only made this example to show how Scrapling features can be used to scrape a
""" """
import requests import requests
from scrapling import Adaptor from scrapling import Adaptor
response = requests.get('https://stackoverflow.com/questions/tagged/web-scraping?sort=MostVotes&filters=NoAcceptedAnswer&edited=true&pagesize=50&page=2') response = requests.get('https://stackoverflow.com/questions/tagged/web-scraping?sort=MostVotes&filters=NoAcceptedAnswer&edited=true&pagesize=50&page=2')
@@ -22,4 +23,3 @@ if first_question_title and first_question_author:
# We will get all the rest of the titles/authors in the page depending on the first title and the first author we got above as a starting point # We will get all the rest of the titles/authors in the page depending on the first title and the first author we got above as a starting point
for i, (title, author) in enumerate(zip(first_question_title.find_similar(), first_question_author.find_similar()), start=1): for i, (title, author) in enumerate(zip(first_question_title.find_similar(), first_question_author.find_similar()), start=1):
print(i, title.text, author.text) print(i, title.text, author.text)
+3 -2
View File
@@ -1,7 +1,8 @@
# Declare top-level shortcuts # Declare top-level shortcuts
from scrapling.fetchers import Fetcher, StealthyFetcher, PlayWrightFetcher, CustomFetcher from scrapling.core.custom_types import AttributesHandler, TextHandler
from scrapling.fetchers import (CustomFetcher, Fetcher, PlayWrightFetcher,
StealthyFetcher)
from scrapling.parser import Adaptor, Adaptors from scrapling.parser import Adaptor, Adaptors
from scrapling.core.custom_types import TextHandler, AttributesHandler
__author__ = "Karim Shoair (karim.shoair@pm.me)" __author__ = "Karim Shoair (karim.shoair@pm.me)"
__version__ = "0.2.7" __version__ = "0.2.7"
+2 -3
View File
@@ -2,9 +2,8 @@
Type definitions for type checking purposes. Type definitions for type checking purposes.
""" """
from typing import ( from typing import (TYPE_CHECKING, Any, Callable, Dict, Generator, Iterable,
Dict, Optional, Union, Callable, Any, List, Tuple, Pattern, Generator, Iterable, Type, TYPE_CHECKING, Literal List, Literal, Optional, Pattern, Tuple, Type, Union)
)
try: try:
from typing import Protocol from typing import Protocol
+5 -5
View File
@@ -1,13 +1,13 @@
import re import re
from types import MappingProxyType
from collections.abc import Mapping from collections.abc import Mapping
from types import MappingProxyType
from scrapling.core.utils import _is_iterable, flatten from orjson import dumps, loads
from scrapling.core._types import Dict, List, Union, Pattern, SupportsIndex
from orjson import loads, dumps
from w3lib.html import replace_entities as _replace_entities from w3lib.html import replace_entities as _replace_entities
from scrapling.core._types import Dict, List, Pattern, SupportsIndex, Union
from scrapling.core.utils import _is_iterable, flatten
class TextHandler(str): class TextHandler(str):
"""Extends standard Python string by adding more functionality""" """Extends standard Python string by adding more functionality"""
+6 -6
View File
@@ -1,16 +1,16 @@
import orjson
import sqlite3
import logging import logging
import sqlite3
import threading import threading
from hashlib import sha256
from abc import ABC, abstractmethod from abc import ABC, abstractmethod
from hashlib import sha256
import orjson
from lxml import html
from tldextract import extract as tld
from scrapling.core._types import Dict, Optional, Union from scrapling.core._types import Dict, Optional, Union
from scrapling.core.utils import _StorageTools, cache from scrapling.core.utils import _StorageTools, cache
from lxml import html
from tldextract import extract as tld
class StorageSystemMixin(ABC): class StorageSystemMixin(ABC):
# If you want to make your own storage system, you have to inherit from this # If you want to make your own storage system, you have to inherit from this
+5 -6
View File
@@ -10,15 +10,14 @@ So you don't have to learn a new selectors/api method like what bs4 done with so
import re import re
from w3lib.html import HTML5_WHITESPACE
from scrapling.core.utils import cache
from scrapling.core._types import Any, Optional, Protocol, Self
from cssselect.xpath import ExpressionError
from cssselect.xpath import XPathExpr as OriginalXPathExpr
from cssselect import HTMLTranslator as OriginalHTMLTranslator from cssselect import HTMLTranslator as OriginalHTMLTranslator
from cssselect.parser import Element, FunctionalPseudoElement, PseudoElement from cssselect.parser import Element, FunctionalPseudoElement, PseudoElement
from cssselect.xpath import ExpressionError
from cssselect.xpath import XPathExpr as OriginalXPathExpr
from w3lib.html import HTML5_WHITESPACE
from scrapling.core._types import Any, Optional, Protocol, Self
from scrapling.core.utils import cache
regex = f"[{HTML5_WHITESPACE}]+" regex = f"[{HTML5_WHITESPACE}]+"
replace_html5_whitespaces = re.compile(regex).sub replace_html5_whitespaces = re.compile(regex).sub
+8 -5
View File
@@ -1,14 +1,17 @@
import re
import logging import logging
import re
from itertools import chain from itertools import chain
# Using cache on top of a class is brilliant way to achieve Singleton design pattern without much code
from functools import lru_cache as cache # functools.cache is available on Python 3.9+ only so let's keep lru_cache
from scrapling.core._types import Dict, Iterable, Any, Union
import orjson import orjson
from lxml import html from lxml import html
from scrapling.core._types import Any, Dict, Iterable, Union
# Using cache on top of a class is brilliant way to achieve Singleton design pattern without much code
# functools.cache is available on Python 3.9+ only so let's keep lru_cache
from functools import lru_cache as cache # isort:skip
html_forbidden = {html.HtmlComment, } html_forbidden = {html.HtmlComment, }
logging.basicConfig( logging.basicConfig(
level=logging.ERROR, level=logging.ERROR,
+1 -1
View File
@@ -1,4 +1,4 @@
from .fetchers import Fetcher, StealthyFetcher, PlayWrightFetcher from .fetchers import Fetcher, PlayWrightFetcher, StealthyFetcher
# If you are going to use Fetchers with the default settings, import them from this file instead for a cleaner looking code # If you are going to use Fetchers with the default settings, import them from this file instead for a cleaner looking code
Fetcher = Fetcher() Fetcher = Fetcher()
+2 -2
View File
@@ -1,7 +1,7 @@
from .camo import CamoufoxEngine from .camo import CamoufoxEngine
from .static import StaticEngine
from .pw import PlaywrightEngine
from .constants import DEFAULT_DISABLED_RESOURCES, DEFAULT_STEALTH_FLAGS from .constants import DEFAULT_DISABLED_RESOURCES, DEFAULT_STEALTH_FLAGS
from .pw import PlaywrightEngine
from .static import StaticEngine
from .toolbelt import check_if_engine_usable from .toolbelt import check_if_engine_usable
__all__ = ['CamoufoxEngine', 'PlaywrightEngine'] __all__ = ['CamoufoxEngine', 'PlaywrightEngine']
+8 -12
View File
@@ -1,20 +1,16 @@
import logging import logging
from scrapling.core._types import Union, Callable, Optional, Dict, List, Literal
from scrapling.engines.toolbelt import (
Response,
do_nothing,
StatusText,
get_os_name,
intercept_route,
check_type_validity,
construct_proxy_dict,
generate_convincing_referer,
)
from camoufox import DefaultAddons from camoufox import DefaultAddons
from camoufox.sync_api import Camoufox from camoufox.sync_api import Camoufox
from scrapling.core._types import (Callable, Dict, List, Literal, Optional,
Union)
from scrapling.engines.toolbelt import (Response, StatusText,
check_type_validity,
construct_proxy_dict, do_nothing,
generate_convincing_referer,
get_os_name, intercept_route)
class CamoufoxEngine: class CamoufoxEngine:
def __init__( def __init__(
+9 -14
View File
@@ -1,20 +1,15 @@
import json import json
import logging import logging
from scrapling.core._types import Union, Callable, Optional, List, Dict
from scrapling.engines.constants import DEFAULT_STEALTH_FLAGS, NSTBROWSER_DEFAULT_QUERY from scrapling.core._types import Callable, Dict, List, Optional, Union
from scrapling.engines.toolbelt import ( from scrapling.engines.constants import (DEFAULT_STEALTH_FLAGS,
Response, NSTBROWSER_DEFAULT_QUERY)
do_nothing, from scrapling.engines.toolbelt import (Response, StatusText,
StatusText, check_type_validity, construct_cdp_url,
js_bypass_path, construct_proxy_dict, do_nothing,
intercept_route, generate_convincing_referer,
generate_headers, generate_headers, intercept_route,
construct_cdp_url, js_bypass_path)
check_type_validity,
construct_proxy_dict,
generate_convincing_referer,
)
class PlaywrightEngine: class PlaywrightEngine:
+4 -3
View File
@@ -1,11 +1,12 @@
import logging import logging
from scrapling.core._types import Union, Optional, Dict
from .toolbelt import Response, generate_convincing_referer, generate_headers
import httpx import httpx
from httpx._models import Response as httpxResponse from httpx._models import Response as httpxResponse
from scrapling.core._types import Dict, Optional, Union
from .toolbelt import Response, generate_convincing_referer, generate_headers
class StaticEngine: class StaticEngine:
def __init__(self, follow_redirects: bool = True, timeout: Optional[Union[int, float]] = None, adaptor_arguments: Dict = None): def __init__(self, follow_redirects: bool = True, timeout: Optional[Union[int, float]] = None, adaptor_arguments: Dict = None):
+6 -20
View File
@@ -1,20 +1,6 @@
from .fingerprints import ( from .custom import (BaseFetcher, Response, StatusText, check_if_engine_usable,
get_os_name, check_type_validity, do_nothing, get_variable_name)
generate_headers, from .fingerprints import (generate_convincing_referer, generate_headers,
generate_convincing_referer, get_os_name)
) from .navigation import (construct_cdp_url, construct_proxy_dict,
from .custom import ( intercept_route, js_bypass_path)
Response,
do_nothing,
StatusText,
BaseFetcher,
get_variable_name,
check_type_validity,
check_if_engine_usable,
)
from .navigation import (
js_bypass_path,
intercept_route,
construct_cdp_url,
construct_proxy_dict,
)
+3 -2
View File
@@ -5,10 +5,11 @@ import inspect
import logging import logging
from email.message import Message from email.message import Message
from scrapling.core._types import (Any, Callable, Dict, List, Optional, Tuple,
Type, Union)
from scrapling.core.custom_types import MappingProxyType from scrapling.core.custom_types import MappingProxyType
from scrapling.core.utils import cache, setup_basic_logging
from scrapling.parser import Adaptor, SQLiteStorageSystem from scrapling.parser import Adaptor, SQLiteStorageSystem
from scrapling.core.utils import setup_basic_logging, cache
from scrapling.core._types import Any, List, Type, Union, Optional, Dict, Callable, Tuple
class ResponseEncoding: class ResponseEncoding:
+5 -5
View File
@@ -4,12 +4,12 @@ Functions related to generating headers and fingerprints generally
import platform import platform
from scrapling.core.utils import cache from browserforge.fingerprints import Fingerprint, FingerprintGenerator
from scrapling.core._types import Union, Dict from browserforge.headers import Browser, HeaderGenerator
from tldextract import extract from tldextract import extract
from browserforge.headers import HeaderGenerator, Browser
from browserforge.fingerprints import FingerprintGenerator, Fingerprint from scrapling.core._types import Dict, Union
from scrapling.core.utils import cache
@cache(None, typed=True) @cache(None, typed=True)
+6 -6
View File
@@ -2,16 +2,16 @@
Functions related to files and URLs Functions related to files and URLs
""" """
import os
import logging import logging
from urllib.parse import urlparse, urlencode import os
from urllib.parse import urlencode, urlparse
from scrapling.core.utils import cache
from scrapling.core._types import Union, Dict, Optional
from scrapling.engines.constants import DEFAULT_DISABLED_RESOURCES
from playwright.sync_api import Route from playwright.sync_api import Route
from scrapling.core._types import Dict, Optional, Union
from scrapling.core.utils import cache
from scrapling.engines.constants import DEFAULT_DISABLED_RESOURCES
def intercept_route(route: Route) -> Union[Route, None]: def intercept_route(route: Route) -> Union[Route, None]:
"""This is just a route handler but it drops requests that its type falls in `DEFAULT_DISABLED_RESOURCES` """This is just a route handler but it drops requests that its type falls in `DEFAULT_DISABLED_RESOURCES`
+5 -4
View File
@@ -1,7 +1,8 @@
from scrapling.core._types import Dict, Optional, Union, Callable, List, Literal from scrapling.core._types import (Callable, Dict, List, Literal, Optional,
Union)
from scrapling.engines.toolbelt import Response, BaseFetcher, do_nothing from scrapling.engines import (CamoufoxEngine, PlaywrightEngine, StaticEngine,
from scrapling.engines import CamoufoxEngine, PlaywrightEngine, StaticEngine, check_if_engine_usable check_if_engine_usable)
from scrapling.engines.toolbelt import BaseFetcher, Response, do_nothing
class Fetcher(BaseFetcher): class Fetcher(BaseFetcher):
+15 -8
View File
@@ -1,16 +1,23 @@
import inspect
import os import os
import re import re
import inspect
from difflib import SequenceMatcher from difflib import SequenceMatcher
from scrapling.core.translator import HTMLTranslator from cssselect import SelectorError, SelectorSyntaxError
from scrapling.core.mixins import SelectorsGeneration from cssselect import parse as split_selectors
from scrapling.core.custom_types import TextHandler, TextHandlers, AttributesHandler
from scrapling.core.storage_adaptors import SQLiteStorageSystem, StorageSystemMixin, _StorageTools
from scrapling.core.utils import setup_basic_logging, logging, clean_spaces, flatten, html_forbidden, is_jsonable
from scrapling.core._types import Any, Dict, List, Tuple, Optional, Pattern, Union, Callable, Generator, SupportsIndex, Iterable
from lxml import etree, html from lxml import etree, html
from cssselect import SelectorError, SelectorSyntaxError, parse as split_selectors
from scrapling.core._types import (Any, Callable, Dict, Generator, Iterable,
List, Optional, Pattern, SupportsIndex,
Tuple, Union)
from scrapling.core.custom_types import (AttributesHandler, TextHandler,
TextHandlers)
from scrapling.core.mixins import SelectorsGeneration
from scrapling.core.storage_adaptors import (SQLiteStorageSystem,
StorageSystemMixin, _StorageTools)
from scrapling.core.translator import HTMLTranslator
from scrapling.core.utils import (clean_spaces, flatten, html_forbidden,
is_jsonable, logging, setup_basic_logging)
class Adaptor(SelectorsGeneration): class Adaptor(SelectorsGeneration):
+1 -1
View File
@@ -1,4 +1,4 @@
from setuptools import setup, find_packages from setuptools import find_packages, setup
with open("README.md", "r", encoding="utf-8") as fh: with open("README.md", "r", encoding="utf-8") as fh:
long_description = fh.read() long_description = fh.read()
+1
View File
@@ -1,4 +1,5 @@
import unittest import unittest
import pytest_httpbin import pytest_httpbin
from scrapling import StealthyFetcher from scrapling import StealthyFetcher
+1
View File
@@ -1,4 +1,5 @@
import unittest import unittest
import pytest_httpbin import pytest_httpbin
from scrapling import Fetcher from scrapling import Fetcher
+1
View File
@@ -1,4 +1,5 @@
import unittest import unittest
import pytest_httpbin import pytest_httpbin
from scrapling import PlayWrightFetcher from scrapling import PlayWrightFetcher
+3 -1
View File
@@ -1,9 +1,11 @@
import pickle import pickle
import unittest import unittest
from scrapling import Adaptor
from cssselect import SelectorError, SelectorSyntaxError from cssselect import SelectorError, SelectorSyntaxError
from scrapling import Adaptor
class TestParser(unittest.TestCase): class TestParser(unittest.TestCase):
def setUp(self): def setUp(self):