Return to the default text behaviour
The logic to remove comment tags before returning text was causing errors sometimes so from now on leave `keep_comments` set to True better as it is.
This commit is contained in:
+2
-16
@@ -187,22 +187,8 @@ class Adaptor(SelectorsGeneration):
|
|||||||
def text(self) -> TextHandler:
|
def text(self) -> TextHandler:
|
||||||
"""Get text content of the element"""
|
"""Get text content of the element"""
|
||||||
if not self.__text:
|
if not self.__text:
|
||||||
if self.__keep_comments:
|
# If you want to escape lxml default behaviour and remove comments like this `<span>CONDITION: <!-- -->Excellent</span>`
|
||||||
if not self.children:
|
# before extracting text then keep `keep_comments` set to False while initializing the first class
|
||||||
# If use chose to keep comments, remove comments from text
|
|
||||||
# Escape lxml default behaviour and remove comments like this `<span>CONDITION: <!-- -->Excellent</span>`
|
|
||||||
# This issue is present in parsel/scrapy as well so no need to repeat it here so the user can run regex on the full text.
|
|
||||||
code = self.html_content
|
|
||||||
parser = html.HTMLParser(
|
|
||||||
recover=True, remove_blank_text=True, remove_comments=True, encoding=self.encoding,
|
|
||||||
compact=True, huge_tree=self.__huge_tree_enabled, default_doctype=True
|
|
||||||
)
|
|
||||||
fragment_root = html.fragment_fromstring(code, parser=parser)
|
|
||||||
self.__text = TextHandler(fragment_root.text)
|
|
||||||
else:
|
|
||||||
self.__text = TextHandler(self._root.text)
|
|
||||||
else:
|
|
||||||
# If user already chose to not keep comments then all is good
|
|
||||||
self.__text = TextHandler(self._root.text)
|
self.__text = TextHandler(self._root.text)
|
||||||
return self.__text
|
return self.__text
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user