Parser - New logic to handle JSON responses passed from Fetchers
This commit is contained in:
+17
-7
@@ -10,6 +10,7 @@ from scrapling.core.storage_adaptors import SQLiteStorageSystem, StorageSystemMi
|
|||||||
from scrapling.core.utils import setup_basic_logging, logging, clean_spaces, flatten, html_forbidden
|
from scrapling.core.utils import setup_basic_logging, logging, clean_spaces, flatten, html_forbidden
|
||||||
from scrapling.core._types import Any, Dict, List, Tuple, Optional, Pattern, Union, Callable, Generator, SupportsIndex, Iterable
|
from scrapling.core._types import Any, Dict, List, Tuple, Optional, Pattern, Union, Callable, Generator, SupportsIndex, Iterable
|
||||||
from lxml import etree, html
|
from lxml import etree, html
|
||||||
|
from lxml.etree import XMLSyntaxError
|
||||||
from cssselect import SelectorError, SelectorSyntaxError, parse as split_selectors
|
from cssselect import SelectorError, SelectorSyntaxError, parse as split_selectors
|
||||||
|
|
||||||
|
|
||||||
@@ -60,6 +61,7 @@ class Adaptor(SelectorsGeneration):
|
|||||||
if root is None and not body and text is None:
|
if root is None and not body and text is None:
|
||||||
raise ValueError("Adaptor class needs text, body, or root arguments to work")
|
raise ValueError("Adaptor class needs text, body, or root arguments to work")
|
||||||
|
|
||||||
|
self.__text = None
|
||||||
if root is None:
|
if root is None:
|
||||||
if text is None:
|
if text is None:
|
||||||
if not body or not isinstance(body, bytes):
|
if not body or not isinstance(body, bytes):
|
||||||
@@ -72,12 +74,21 @@ class Adaptor(SelectorsGeneration):
|
|||||||
|
|
||||||
body = text.strip().replace("\x00", "").encode(encoding) or b"<html/>"
|
body = text.strip().replace("\x00", "").encode(encoding) or b"<html/>"
|
||||||
|
|
||||||
parser = html.HTMLParser(
|
# https://lxml.de/api/lxml.etree.HTMLParser-class.html
|
||||||
# https://lxml.de/api/lxml.etree.HTMLParser-class.html
|
try:
|
||||||
recover=True, remove_blank_text=True, remove_comments=(keep_comments is False), encoding=encoding,
|
# Test with recover set to False first so if this is a text body like a json response, we get error
|
||||||
compact=True, huge_tree=huge_tree, default_doctype=True
|
parser = html.HTMLParser(
|
||||||
)
|
recover=False, remove_blank_text=True, remove_comments=(keep_comments is False), encoding=encoding,
|
||||||
self._root = etree.fromstring(body, parser=parser, base_url=url)
|
compact=True, huge_tree=huge_tree, default_doctype=True
|
||||||
|
)
|
||||||
|
self._root = etree.fromstring(body, parser=parser, base_url=url)
|
||||||
|
except XMLSyntaxError:
|
||||||
|
parser = html.HTMLParser(
|
||||||
|
recover=True, remove_blank_text=True, remove_comments=(keep_comments is False), encoding=encoding,
|
||||||
|
compact=True, huge_tree=huge_tree, default_doctype=True
|
||||||
|
)
|
||||||
|
self._root = etree.fromstring(body, parser=parser, base_url=url)
|
||||||
|
self.__text = TextHandler(text or body.decode())
|
||||||
|
|
||||||
else:
|
else:
|
||||||
# All html types inherits from HtmlMixin so this to check for all at once
|
# All html types inherits from HtmlMixin so this to check for all at once
|
||||||
@@ -112,7 +123,6 @@ class Adaptor(SelectorsGeneration):
|
|||||||
self.url = url
|
self.url = url
|
||||||
# For selector stuff
|
# For selector stuff
|
||||||
self.__attributes = None
|
self.__attributes = None
|
||||||
self.__text = None
|
|
||||||
self.__tag = None
|
self.__tag = None
|
||||||
self.__debug = debug
|
self.__debug = debug
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user