Version 0.1.2

Check the changelog in releases
This commit is contained in:
Karim shoair
2024-10-16 18:21:56 +03:00
parent f0a21f6f38
commit a01cfe66db
7 changed files with 25 additions and 7 deletions
+2
View File
@@ -1,4 +1,6 @@
include LICENSE include LICENSE
include *.db
include scrapling/*.db
include scrapling/py.typed include scrapling/py.typed
recursive-exclude * __pycache__ recursive-exclude * __pycache__
+4 -1
View File
@@ -1,5 +1,5 @@
# 🕷️ Scrapling: Lightning-Fast, Adaptive Web Scraping for Python # 🕷️ Scrapling: Lightning-Fast, Adaptive Web Scraping for Python
[![Tests](https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml/badge.svg)](https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml) [![PyPI version](https://badge.fury.io/py/Scrapling.svg)](https://badge.fury.io/py/Scrapling) [![Supported Python versions](https://img.shields.io/pypi/pyversions/scrapling.svg)](https://pypi.org/project/scrapling/) [![License](https://img.shields.io/badge/License-BSD--3-blue.svg)](https://opensource.org/licenses/BSD-3-Clause) [![Tests](https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml/badge.svg)](https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml) [![PyPI version](https://badge.fury.io/py/Scrapling.svg)](https://badge.fury.io/py/Scrapling) [![Supported Python versions](https://img.shields.io/pypi/pyversions/scrapling.svg)](https://pypi.org/project/scrapling/) [![PyPI Downloads](https://static.pepy.tech/badge/scrapling)](https://pepy.tech/project/scrapling)
Dealing with failing web scrapers due to website changes? Meet Scrapling. Dealing with failing web scrapers due to website changes? Meet Scrapling.
@@ -415,6 +415,9 @@ Of course, you can find elements by text/regex, find similar elements in a more
### Is Scrapling thread-safe? ### Is Scrapling thread-safe?
Yes, Scrapling instances are thread-safe. Each Adaptor instance maintains its own state. Yes, Scrapling instances are thread-safe. Each Adaptor instance maintains its own state.
## Sponsors
[![Capsolver Banner](https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/CapSolver.png)](https://www.capsolver.com/?utm_source=github&utm_medium=repo&utm_campaign=scraping&utm_term=Scrapling)
## Contributing ## Contributing
Everybody is invited and welcome to contribute to Scrapling. There is a lot to do! Everybody is invited and welcome to contribute to Scrapling. There is a lot to do!
Binary file not shown.

After

Width:  |  Height:  |  Size: 173 KiB

+1 -1
View File
@@ -3,7 +3,7 @@ from scrapling.parser import Adaptor, Adaptors
from scrapling.custom_types import TextHandler, AttributesHandler from scrapling.custom_types import TextHandler, AttributesHandler
__author__ = "Karim Shoair (karim.shoair@pm.me)" __author__ = "Karim Shoair (karim.shoair@pm.me)"
__version__ = "0.1.1" __version__ = "0.1.2"
__copyright__ = "Copyright (c) 2024 Karim Shoair" __copyright__ = "Copyright (c) 2024 Karim Shoair"
+15 -2
View File
@@ -78,7 +78,7 @@ class Adaptor(SelectorsGeneration):
parser = html.HTMLParser( parser = html.HTMLParser(
# https://lxml.de/api/lxml.etree.HTMLParser-class.html # https://lxml.de/api/lxml.etree.HTMLParser-class.html
recover=True, remove_blank_text=True, remove_comments=(keep_comments is True), encoding=encoding, recover=True, remove_blank_text=True, remove_comments=(keep_comments is False), encoding=encoding,
compact=True, huge_tree=huge_tree, default_doctype=True compact=True, huge_tree=huge_tree, default_doctype=True
) )
self._root = etree.fromstring(body, parser=parser, base_url=url) self._root = etree.fromstring(body, parser=parser, base_url=url)
@@ -142,7 +142,8 @@ class Adaptor(SelectorsGeneration):
if issubclass(type(element), html.HtmlMixin): if issubclass(type(element), html.HtmlMixin):
return self.__class__( return self.__class__(
root=element, url=self.url, encoding=self.encoding, auto_match=self.__auto_match_enabled, root=element, url=self.url, encoding=self.encoding, auto_match=self.__auto_match_enabled,
keep_comments=self.__keep_comments, huge_tree=self.__huge_tree_enabled, debug=self.__debug keep_comments=True, # if the comments are already removed in initialization, no need to try to delete them in sub-elements
huge_tree=self.__huge_tree_enabled, debug=self.__debug
) )
return element return element
@@ -186,6 +187,18 @@ class Adaptor(SelectorsGeneration):
def text(self) -> TextHandler: def text(self) -> TextHandler:
"""Get text content of the element""" """Get text content of the element"""
if not self.__text: if not self.__text:
if self.__keep_comments:
# If use chose to keep comments, remove comments from text
# Escape lxml default behaviour and remove comments like this `<span>CONDITION: <!-- -->Excellent</span>`
# This issue is present in parsel/scrapy as well so no need to repeat it here so the user can run regex on the full text.
code = self.html_content
parser = html.HTMLParser(
recover=True, remove_blank_text=True, remove_comments=True, encoding=self.encoding,
compact=True, huge_tree=self.__huge_tree_enabled, default_doctype=True
)
fragment_root = html.fragment_fromstring(code, parser=parser)
self.__text = TextHandler(fragment_root.text)
else:
self.__text = TextHandler(self._root.text) self.__text = TextHandler(self._root.text)
return self.__text return self.__text
+1 -1
View File
@@ -1,6 +1,6 @@
[metadata] [metadata]
name = scrapling name = scrapling
version = 0.1.1 version = 0.1.2
author = Karim Shoair author = Karim Shoair
author_email = karim.shoair@pm.me author_email = karim.shoair@pm.me
description = Scrapling is a powerful, flexible, adaptive, and high-performance web scraping library for Python. description = Scrapling is a powerful, flexible, adaptive, and high-performance web scraping library for Python.
+1 -1
View File
@@ -6,7 +6,7 @@ with open("README.md", "r", encoding="utf-8") as fh:
setup( setup(
name="scrapling", name="scrapling",
version="0.1.1", version="0.1.2",
description="""Scrapling is a powerful, flexible, and high-performance web scraping library for Python. It description="""Scrapling is a powerful, flexible, and high-performance web scraping library for Python. It
simplifies the process of extracting data from websites, even when they undergo structural changes, and offers simplifies the process of extracting data from websites, even when they undergo structural changes, and offers
impressive speed improvements over many popular scraping tools.""", impressive speed improvements over many popular scraping tools.""",