Better logic to handle json responses

This commit is contained in:
Karim shoair
2024-11-14 00:03:26 +02:00
parent d4b896b4e9
commit eaa7da27c6
2 changed files with 20 additions and 16 deletions
+13 -1
View File
@@ -4,8 +4,9 @@ from itertools import chain
# Using cache on top of a class is brilliant way to achieve Singleton design pattern without much code # Using cache on top of a class is brilliant way to achieve Singleton design pattern without much code
from functools import lru_cache as cache # functools.cache is available on Python 3.9+ only so let's keep lru_cache from functools import lru_cache as cache # functools.cache is available on Python 3.9+ only so let's keep lru_cache
from scrapling.core._types import Dict, Iterable, Any from scrapling.core._types import Dict, Iterable, Any, Union
import orjson
from lxml import html from lxml import html
html_forbidden = {html.HtmlComment, } html_forbidden = {html.HtmlComment, }
@@ -18,6 +19,17 @@ logging.basicConfig(
) )
def is_jsonable(content: Union[bytes, str]) -> bool:
if type(content) is bytes:
content = content.decode()
try:
_ = orjson.loads(content)
return True
except orjson.JSONDecodeError:
return False
@cache(None, typed=True) @cache(None, typed=True)
def setup_basic_logging(level: str = 'debug'): def setup_basic_logging(level: str = 'debug'):
levels = { levels = {
+7 -15
View File
@@ -7,10 +7,9 @@ from scrapling.core.translator import HTMLTranslator
from scrapling.core.mixins import SelectorsGeneration from scrapling.core.mixins import SelectorsGeneration
from scrapling.core.custom_types import TextHandler, TextHandlers, AttributesHandler from scrapling.core.custom_types import TextHandler, TextHandlers, AttributesHandler
from scrapling.core.storage_adaptors import SQLiteStorageSystem, StorageSystemMixin, _StorageTools from scrapling.core.storage_adaptors import SQLiteStorageSystem, StorageSystemMixin, _StorageTools
from scrapling.core.utils import setup_basic_logging, logging, clean_spaces, flatten, html_forbidden from scrapling.core.utils import setup_basic_logging, logging, clean_spaces, flatten, html_forbidden, is_jsonable
from scrapling.core._types import Any, Dict, List, Tuple, Optional, Pattern, Union, Callable, Generator, SupportsIndex, Iterable from scrapling.core._types import Any, Dict, List, Tuple, Optional, Pattern, Union, Callable, Generator, SupportsIndex, Iterable
from lxml import etree, html from lxml import etree, html
from lxml.etree import XMLSyntaxError
from cssselect import SelectorError, SelectorSyntaxError, parse as split_selectors from cssselect import SelectorError, SelectorSyntaxError, parse as split_selectors
@@ -75,19 +74,12 @@ class Adaptor(SelectorsGeneration):
body = text.strip().replace("\x00", "").encode(encoding) or b"<html/>" body = text.strip().replace("\x00", "").encode(encoding) or b"<html/>"
# https://lxml.de/api/lxml.etree.HTMLParser-class.html # https://lxml.de/api/lxml.etree.HTMLParser-class.html
try: parser = html.HTMLParser(
# Test with recover set to False first so if this is a text body like a json response, we get error recover=True, remove_blank_text=True, remove_comments=(keep_comments is False), encoding=encoding,
parser = html.HTMLParser( compact=True, huge_tree=huge_tree, default_doctype=True
recover=False, remove_blank_text=True, remove_comments=(keep_comments is False), encoding=encoding, )
compact=True, huge_tree=huge_tree, default_doctype=True self._root = etree.fromstring(body, parser=parser, base_url=url)
) if is_jsonable(text or body.decode()):
self._root = etree.fromstring(body, parser=parser, base_url=url)
except XMLSyntaxError:
parser = html.HTMLParser(
recover=True, remove_blank_text=True, remove_comments=(keep_comments is False), encoding=encoding,
compact=True, huge_tree=huge_tree, default_doctype=True
)
self._root = etree.fromstring(body, parser=parser, base_url=url)
self.__text = TextHandler(text or body.decode()) self.__text = TextHandler(text or body.decode())
else: else: