Better logic to handle json responses
This commit is contained in:
+13
-1
@@ -4,8 +4,9 @@ from itertools import chain
|
|||||||
# Using cache on top of a class is brilliant way to achieve Singleton design pattern without much code
|
# Using cache on top of a class is brilliant way to achieve Singleton design pattern without much code
|
||||||
from functools import lru_cache as cache # functools.cache is available on Python 3.9+ only so let's keep lru_cache
|
from functools import lru_cache as cache # functools.cache is available on Python 3.9+ only so let's keep lru_cache
|
||||||
|
|
||||||
from scrapling.core._types import Dict, Iterable, Any
|
from scrapling.core._types import Dict, Iterable, Any, Union
|
||||||
|
|
||||||
|
import orjson
|
||||||
from lxml import html
|
from lxml import html
|
||||||
|
|
||||||
html_forbidden = {html.HtmlComment, }
|
html_forbidden = {html.HtmlComment, }
|
||||||
@@ -18,6 +19,17 @@ logging.basicConfig(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def is_jsonable(content: Union[bytes, str]) -> bool:
|
||||||
|
if type(content) is bytes:
|
||||||
|
content = content.decode()
|
||||||
|
|
||||||
|
try:
|
||||||
|
_ = orjson.loads(content)
|
||||||
|
return True
|
||||||
|
except orjson.JSONDecodeError:
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
@cache(None, typed=True)
|
@cache(None, typed=True)
|
||||||
def setup_basic_logging(level: str = 'debug'):
|
def setup_basic_logging(level: str = 'debug'):
|
||||||
levels = {
|
levels = {
|
||||||
|
|||||||
+7
-15
@@ -7,10 +7,9 @@ from scrapling.core.translator import HTMLTranslator
|
|||||||
from scrapling.core.mixins import SelectorsGeneration
|
from scrapling.core.mixins import SelectorsGeneration
|
||||||
from scrapling.core.custom_types import TextHandler, TextHandlers, AttributesHandler
|
from scrapling.core.custom_types import TextHandler, TextHandlers, AttributesHandler
|
||||||
from scrapling.core.storage_adaptors import SQLiteStorageSystem, StorageSystemMixin, _StorageTools
|
from scrapling.core.storage_adaptors import SQLiteStorageSystem, StorageSystemMixin, _StorageTools
|
||||||
from scrapling.core.utils import setup_basic_logging, logging, clean_spaces, flatten, html_forbidden
|
from scrapling.core.utils import setup_basic_logging, logging, clean_spaces, flatten, html_forbidden, is_jsonable
|
||||||
from scrapling.core._types import Any, Dict, List, Tuple, Optional, Pattern, Union, Callable, Generator, SupportsIndex, Iterable
|
from scrapling.core._types import Any, Dict, List, Tuple, Optional, Pattern, Union, Callable, Generator, SupportsIndex, Iterable
|
||||||
from lxml import etree, html
|
from lxml import etree, html
|
||||||
from lxml.etree import XMLSyntaxError
|
|
||||||
from cssselect import SelectorError, SelectorSyntaxError, parse as split_selectors
|
from cssselect import SelectorError, SelectorSyntaxError, parse as split_selectors
|
||||||
|
|
||||||
|
|
||||||
@@ -75,19 +74,12 @@ class Adaptor(SelectorsGeneration):
|
|||||||
body = text.strip().replace("\x00", "").encode(encoding) or b"<html/>"
|
body = text.strip().replace("\x00", "").encode(encoding) or b"<html/>"
|
||||||
|
|
||||||
# https://lxml.de/api/lxml.etree.HTMLParser-class.html
|
# https://lxml.de/api/lxml.etree.HTMLParser-class.html
|
||||||
try:
|
parser = html.HTMLParser(
|
||||||
# Test with recover set to False first so if this is a text body like a json response, we get error
|
recover=True, remove_blank_text=True, remove_comments=(keep_comments is False), encoding=encoding,
|
||||||
parser = html.HTMLParser(
|
compact=True, huge_tree=huge_tree, default_doctype=True
|
||||||
recover=False, remove_blank_text=True, remove_comments=(keep_comments is False), encoding=encoding,
|
)
|
||||||
compact=True, huge_tree=huge_tree, default_doctype=True
|
self._root = etree.fromstring(body, parser=parser, base_url=url)
|
||||||
)
|
if is_jsonable(text or body.decode()):
|
||||||
self._root = etree.fromstring(body, parser=parser, base_url=url)
|
|
||||||
except XMLSyntaxError:
|
|
||||||
parser = html.HTMLParser(
|
|
||||||
recover=True, remove_blank_text=True, remove_comments=(keep_comments is False), encoding=encoding,
|
|
||||||
compact=True, huge_tree=huge_tree, default_doctype=True
|
|
||||||
)
|
|
||||||
self._root = etree.fromstring(body, parser=parser, base_url=url)
|
|
||||||
self.__text = TextHandler(text or body.decode())
|
self.__text = TextHandler(text or body.decode())
|
||||||
|
|
||||||
else:
|
else:
|
||||||
|
|||||||
Reference in New Issue
Block a user