From 572df6b3b48955ffac484e4fa3691a9b41838f8e Mon Sep 17 00:00:00 2001 From: Karim shoair Date: Sat, 16 Nov 2024 13:54:55 +0200 Subject: [PATCH 1/6] Fixing the way `Response` object handles sub items in some edge cases + Do it with the least impact on performance --- scrapling/engines/camo.py | 4 ++-- scrapling/engines/pw.py | 4 ++-- scrapling/engines/static.py | 4 ++-- scrapling/engines/toolbelt/custom.py | 5 ++--- scrapling/parser.py | 13 +++++++++++-- 5 files changed, 19 insertions(+), 11 deletions(-) diff --git a/scrapling/engines/camo.py b/scrapling/engines/camo.py index 4131151..62fdd39 100644 --- a/scrapling/engines/camo.py +++ b/scrapling/engines/camo.py @@ -114,14 +114,14 @@ class CamoufoxEngine: response = Response( url=res.url, text=page.content(), - content=res.body(), + body=res.body(), status=res.status, reason=res.status_text, encoding=encoding, cookies={cookie['name']: cookie['value'] for cookie in page.context.cookies()}, headers=res.all_headers(), request_headers=res.request.all_headers(), - adaptor_arguments=self.adaptor_arguments + **self.adaptor_arguments ) page.close() diff --git a/scrapling/engines/pw.py b/scrapling/engines/pw.py index 6c80900..3395ece 100644 --- a/scrapling/engines/pw.py +++ b/scrapling/engines/pw.py @@ -224,14 +224,14 @@ class PlaywrightEngine: response = Response( url=res.url, text=page.content(), - content=res.body(), + body=res.body(), status=res.status, reason=res.status_text, encoding=encoding, cookies={cookie['name']: cookie['value'] for cookie in page.context.cookies()}, headers=res.all_headers(), request_headers=res.request.all_headers(), - adaptor_arguments=self.adaptor_arguments + **self.adaptor_arguments ) page.close() return response diff --git a/scrapling/engines/static.py b/scrapling/engines/static.py index d424752..c106332 100644 --- a/scrapling/engines/static.py +++ b/scrapling/engines/static.py @@ -53,14 +53,14 @@ class StaticEngine: return Response( url=str(response.url), text=response.text, - content=response.content, + body=response.content, status=response.status_code, reason=response.reason_phrase, encoding=response.encoding or 'utf-8', cookies=dict(response.cookies), headers=dict(response.headers), request_headers=dict(response.request.headers), - adaptor_arguments=self.adaptor_arguments + **self.adaptor_arguments ) def get(self, url: str, stealthy_headers: Optional[bool] = True, **kwargs: Dict) -> Response: diff --git a/scrapling/engines/toolbelt/custom.py b/scrapling/engines/toolbelt/custom.py index 274654c..025efda 100644 --- a/scrapling/engines/toolbelt/custom.py +++ b/scrapling/engines/toolbelt/custom.py @@ -12,15 +12,14 @@ from scrapling.core._types import Any, List, Type, Union, Optional, Dict, Callab class Response(Adaptor): """This class is returned by all engines as a way to unify response type between different libraries.""" - def __init__(self, url: str, text: str, content: bytes, status: int, reason: str, cookies: Dict, headers: Dict, request_headers: Dict, adaptor_arguments: Dict, encoding: str = 'utf-8'): + def __init__(self, url: str, text: str, body: bytes, status: int, reason: str, cookies: Dict, headers: Dict, request_headers: Dict, encoding: str = 'utf-8', **adaptor_arguments: Dict): automatch_domain = adaptor_arguments.pop('automatch_domain', None) - super().__init__(text=text, body=content, url=automatch_domain or url, encoding=encoding, **adaptor_arguments) - self.status = status self.reason = reason self.cookies = cookies self.headers = headers self.request_headers = request_headers + super().__init__(text=text, body=body, url=automatch_domain or url, encoding=encoding, **adaptor_arguments) # For back-ward compatibility self.adaptor = self diff --git a/scrapling/parser.py b/scrapling/parser.py index 8d03dda..79cfa14 100644 --- a/scrapling/parser.py +++ b/scrapling/parser.py @@ -32,6 +32,7 @@ class Adaptor(SelectorsGeneration): storage: Any = SQLiteStorageSystem, storage_args: Optional[Dict] = None, debug: Optional[bool] = True, + **kwargs ): """The main class that works as a wrapper for the HTML input data. Using this class, you can search for elements with expressions in CSS, XPath, or with simply text. Check the docs for more info. @@ -117,6 +118,10 @@ class Adaptor(SelectorsGeneration): self.__attributes = None self.__tag = None self.__debug = debug + # No need to check if all response attributes exist or not because if `status` exist, then the rest exist (Save some CPU cycles for speed) + self.__response_data = { + key: getattr(self, key) for key in ('status', 'reason', 'cookies', 'headers', 'request_headers',) + } if hasattr(self, 'status') else {} # Node functionalities, I wanted to move to separate Mixin class but it had slight impact on performance @staticmethod @@ -138,10 +143,14 @@ class Adaptor(SelectorsGeneration): return TextHandler(str(element)) else: if issubclass(type(element), html.HtmlMixin): + return self.__class__( - root=element, url=self.url, encoding=self.encoding, auto_match=self.__auto_match_enabled, + root=element, + text='', body=b'', # Since root argument is provided, both `text` and `body` will be ignored so this is just a filler + url=self.url, encoding=self.encoding, auto_match=self.__auto_match_enabled, keep_comments=True, # if the comments are already removed in initialization, no need to try to delete them in sub-elements - huge_tree=self.__huge_tree_enabled, debug=self.__debug + huge_tree=self.__huge_tree_enabled, debug=self.__debug, + **self.__response_data ) return element From cefc95cc3de53be71f433ce13de826c08aa340e0 Mon Sep 17 00:00:00 2001 From: Karim shoair Date: Sat, 16 Nov 2024 13:57:33 +0200 Subject: [PATCH 2/6] Pump version up to 0.2.2 --- scrapling/__init__.py | 2 +- setup.cfg | 2 +- setup.py | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/scrapling/__init__.py b/scrapling/__init__.py index 4b3b69b..8875c1f 100644 --- a/scrapling/__init__.py +++ b/scrapling/__init__.py @@ -4,7 +4,7 @@ from scrapling.parser import Adaptor, Adaptors from scrapling.core.custom_types import TextHandler, AttributesHandler __author__ = "Karim Shoair (karim.shoair@pm.me)" -__version__ = "0.2.1" +__version__ = "0.2.2" __copyright__ = "Copyright (c) 2024 Karim Shoair" diff --git a/setup.cfg b/setup.cfg index 69c8d16..8c62480 100644 --- a/setup.cfg +++ b/setup.cfg @@ -1,6 +1,6 @@ [metadata] name = scrapling -version = 0.2.1 +version = 0.2.2 author = Karim Shoair author_email = karim.shoair@pm.me description = Scrapling is an undetectable, powerful, flexible, adaptive, and high-performance web scraping library for Python. diff --git a/setup.py b/setup.py index 63340b9..f2197b8 100644 --- a/setup.py +++ b/setup.py @@ -6,7 +6,7 @@ with open("README.md", "r", encoding="utf-8") as fh: setup( name="scrapling", - version="0.2.1", + version="0.2.2", description="""Scrapling is a powerful, flexible, and high-performance web scraping library for Python. It simplifies the process of extracting data from websites, even when they undergo structural changes, and offers impressive speed improvements over many popular scraping tools.""", From 422c135aae359797c11970b3dbb573106061dc16 Mon Sep 17 00:00:00 2001 From: Karim shoair Date: Sat, 16 Nov 2024 21:10:26 +0200 Subject: [PATCH 3/6] Adding a file import initialized fetchers for cleaner looking code and lazy people like me :) --- scrapling/defaults.py | 6 ++++++ 1 file changed, 6 insertions(+) create mode 100644 scrapling/defaults.py diff --git a/scrapling/defaults.py b/scrapling/defaults.py new file mode 100644 index 0000000..79aa2ff --- /dev/null +++ b/scrapling/defaults.py @@ -0,0 +1,6 @@ +from .fetchers import Fetcher, StealthyFetcher, PlayWrightFetcher + +# If you are going to use Fetchers with the default settings, import them from this file instead for a cleaner looking code +Fetcher = Fetcher() +StealthyFetcher = StealthyFetcher() +PlayWrightFetcher = PlayWrightFetcher() From b60ac7f56afc0aa7e4691606e71fb79d6d1f7d9b Mon Sep 17 00:00:00 2001 From: Karim shoair Date: Sat, 16 Nov 2024 21:11:58 +0200 Subject: [PATCH 4/6] Change debug mode to be off by default --- scrapling/engines/toolbelt/custom.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapling/engines/toolbelt/custom.py b/scrapling/engines/toolbelt/custom.py index 025efda..de34099 100644 --- a/scrapling/engines/toolbelt/custom.py +++ b/scrapling/engines/toolbelt/custom.py @@ -30,7 +30,7 @@ class Response(Adaptor): class BaseFetcher: def __init__( self, huge_tree: bool = True, keep_comments: Optional[bool] = False, auto_match: Optional[bool] = True, - storage: Any = SQLiteStorageSystem, storage_args: Optional[Dict] = None, debug: Optional[bool] = True, + storage: Any = SQLiteStorageSystem, storage_args: Optional[Dict] = None, debug: Optional[bool] = False, automatch_domain: Optional[str] = None, ): """Arguments below are the same from the Adaptor class so you can pass them directly, the rest of Adaptor's arguments From d8a8af8d05ee0b040e5e047b2c32d2c84449fbd8 Mon Sep 17 00:00:00 2001 From: Karim shoair Date: Sat, 16 Nov 2024 21:29:05 +0200 Subject: [PATCH 5/6] Updating the README to reflect new changes --- README.md | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index 29e6c6b..34b4068 100644 --- a/README.md +++ b/README.md @@ -6,9 +6,9 @@ Dealing with failing web scrapers due to anti-bot protections or website changes Scrapling is a high-performance, intelligent web scraping library for Python that automatically adapts to website changes while significantly outperforming popular alternatives. For both beginners and experts, Scrapling provides powerful features while maintaining simplicity. ```python ->> from scrapling import Fetcher, StealthyFetcher, PlayWrightFetcher +>> from scrapling.default import Fetcher, StealthyFetcher, PlayWrightFetcher # Fetch websites' source under the radar! ->> page = StealthyFetcher().fetch('https://example.com', headless=True, network_idle=True) +>> page = StealthyFetcher.fetch('https://example.com', headless=True, network_idle=True) >> print(page.status) 200 >> products = page.css('.product', auto_save=True) # Scrape data that survives website design changes! @@ -211,12 +211,21 @@ python -m browserforge update ``` ## Fetching Websites Features -All fetcher-type classes are imported in the same way +You might be a little bit confused by now so let me clear things up. All fetcher-type classes are imported in the same way ```python from scrapling import Fetcher, StealthyFetcher, PlayWrightFetcher ``` And all of them can take these initialization arguments: `auto_match`, `huge_tree`, `keep_comments`, `storage`, `storage_args`, and `debug` which are the same ones you give to the `Adaptor` class. +If you don't want to pass arguments to the generated `Adaptor` object and want to use the default values, you can use import instead for cleaner code: +```python +from scrapling.default import Fetcher, StealthyFetcher, PlayWrightFetcher +``` +then use it right away without initializing like: +```python +page = StealthyFetcher.fetch('https://example.com') +``` + Also, the `Response` object returned from all fetchers is the same as `Adaptor` object except it has these added attributes: `status`, `reason`, `cookies`, `headers`, and `request_headers`. All `cookies`, `headers`, and `request_headers` are always of type `dictionary`. > [!NOTE] > The `auto_match` argument is enabled by default which is the one you should care about the most as you will see later. From 0c6e7708d135fe84549de720731b418fe302a13e Mon Sep 17 00:00:00 2001 From: Karim shoair Date: Sat, 16 Nov 2024 21:55:53 +0200 Subject: [PATCH 6/6] Update README.md --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index 34b4068..028b41d 100644 --- a/README.md +++ b/README.md @@ -217,7 +217,7 @@ from scrapling import Fetcher, StealthyFetcher, PlayWrightFetcher ``` And all of them can take these initialization arguments: `auto_match`, `huge_tree`, `keep_comments`, `storage`, `storage_args`, and `debug` which are the same ones you give to the `Adaptor` class. -If you don't want to pass arguments to the generated `Adaptor` object and want to use the default values, you can use import instead for cleaner code: +If you don't want to pass arguments to the generated `Adaptor` object and want to use the default values, you can use this import instead for cleaner code: ```python from scrapling.default import Fetcher, StealthyFetcher, PlayWrightFetcher ```