Return to the default text behaviour

The logic to remove comment tags before returning text was causing errors sometimes so from now on leave `keep_comments` set to True better as it is.
This commit is contained in:
Karim shoair
2024-11-14 22:21:45 +02:00
parent ea81e0284d
commit b0c2b1818b
+2 -16
View File
@@ -187,22 +187,8 @@ class Adaptor(SelectorsGeneration):
def text(self) -> TextHandler: def text(self) -> TextHandler:
"""Get text content of the element""" """Get text content of the element"""
if not self.__text: if not self.__text:
if self.__keep_comments: # If you want to escape lxml default behaviour and remove comments like this `<span>CONDITION: <!-- -->Excellent</span>`
if not self.children: # before extracting text then keep `keep_comments` set to False while initializing the first class
# If use chose to keep comments, remove comments from text
# Escape lxml default behaviour and remove comments like this `<span>CONDITION: <!-- -->Excellent</span>`
# This issue is present in parsel/scrapy as well so no need to repeat it here so the user can run regex on the full text.
code = self.html_content
parser = html.HTMLParser(
recover=True, remove_blank_text=True, remove_comments=True, encoding=self.encoding,
compact=True, huge_tree=self.__huge_tree_enabled, default_doctype=True
)
fragment_root = html.fragment_fromstring(code, parser=parser)
self.__text = TextHandler(fragment_root.text)
else:
self.__text = TextHandler(self._root.text)
else:
# If user already chose to not keep comments then all is good
self.__text = TextHandler(self._root.text) self.__text = TextHandler(self._root.text)
return self.__text return self.__text