fix: Adjusting cache size in all the library

This commit is contained in:
Karim shoair
2025-03-17 06:46:41 +02:00
parent 18d2d29fa8
commit ff3962b1ad
7 changed files with 12 additions and 12 deletions
+3 -3
View File
@@ -19,7 +19,7 @@ class StorageSystemMixin(ABC):
""" """
self.url = url self.url = url
@lru_cache(126, typed=True) @lru_cache(64, typed=True)
def _get_base_url(self, default_value: str = 'default') -> str: def _get_base_url(self, default_value: str = 'default') -> str:
if not self.url or type(self.url) is not str: if not self.url or type(self.url) is not str:
return default_value return default_value
@@ -51,7 +51,7 @@ class StorageSystemMixin(ABC):
raise NotImplementedError('Storage system must implement `save` method') raise NotImplementedError('Storage system must implement `save` method')
@staticmethod @staticmethod
@lru_cache(256, typed=True) @lru_cache(128, typed=True)
def _get_hash(identifier: str) -> str: def _get_hash(identifier: str) -> str:
"""If you want to hash identifier in your storage system, use this safer""" """If you want to hash identifier in your storage system, use this safer"""
identifier = identifier.lower().strip() identifier = identifier.lower().strip()
@@ -63,7 +63,7 @@ class StorageSystemMixin(ABC):
return f"{hash_value}_{len(identifier)}" # Length to reduce collision chance return f"{hash_value}_{len(identifier)}" # Length to reduce collision chance
@lru_cache(10, typed=True) @lru_cache(1, typed=True)
class SQLiteStorageSystem(StorageSystemMixin): class SQLiteStorageSystem(StorageSystemMixin):
"""The recommended system to use, it's race condition safe and thread safe. """The recommended system to use, it's race condition safe and thread safe.
Mainly built so the library can run in threaded frameworks like scrapy or threaded tools Mainly built so the library can run in threaded frameworks like scrapy or threaded tools
+1 -1
View File
@@ -115,7 +115,7 @@ class _StorageTools:
# return _impl # return _impl
@lru_cache(256, typed=True) @lru_cache(128, typed=True)
def clean_spaces(string): def clean_spaces(string):
string = string.replace('\t', ' ') string = string.replace('\t', ' ')
string = re.sub('[\n|\r]', '', string) string = re.sub('[\n|\r]', '', string)
+2 -2
View File
@@ -126,7 +126,7 @@ class PlaywrightEngine:
return cdp_url return cdp_url
@lru_cache(126, typed=True) @lru_cache(32, typed=True)
def __set_flags(self): def __set_flags(self):
"""Returns the flags that will be used while launching the browser if stealth mode is enabled""" """Returns the flags that will be used while launching the browser if stealth mode is enabled"""
flags = DEFAULT_STEALTH_FLAGS flags = DEFAULT_STEALTH_FLAGS
@@ -169,7 +169,7 @@ class PlaywrightEngine:
return context_kwargs return context_kwargs
@lru_cache(10) @lru_cache(1)
def __stealth_scripts(self): def __stealth_scripts(self):
# Basic bypasses nothing fancy as I'm still working on it # Basic bypasses nothing fancy as I'm still working on it
# But with adding these bypasses to the above config, it bypasses many online tests like # But with adding these bypasses to the above config, it bypasses many online tests like
+1 -1
View File
@@ -7,7 +7,7 @@ from scrapling.core.utils import log, lru_cache
from .toolbelt import Response, generate_convincing_referer, generate_headers from .toolbelt import Response, generate_convincing_referer, generate_headers
@lru_cache(5, typed=True) # Singleton easily @lru_cache(2, typed=True) # Singleton easily
class StaticEngine: class StaticEngine:
def __init__( def __init__(
self, url: str, proxy: Optional[str] = None, stealthy_headers: bool = True, follow_redirects: bool = True, self, url: str, proxy: Optional[str] = None, stealthy_headers: bool = True, follow_redirects: bool = True,
+2 -2
View File
@@ -16,7 +16,7 @@ class ResponseEncoding:
__ISO_8859_1_CONTENT_TYPES = {"text/plain", "text/html", "text/css", "text/javascript"} __ISO_8859_1_CONTENT_TYPES = {"text/plain", "text/html", "text/css", "text/javascript"}
@classmethod @classmethod
@lru_cache(maxsize=256) @lru_cache(maxsize=128)
def __parse_content_type(cls, header_value: str) -> Tuple[str, Dict[str, str]]: def __parse_content_type(cls, header_value: str) -> Tuple[str, Dict[str, str]]:
"""Parse content type and parameters from a content-type header value. """Parse content type and parameters from a content-type header value.
@@ -38,7 +38,7 @@ class ResponseEncoding:
return content_type, params return content_type, params
@classmethod @classmethod
@lru_cache(maxsize=256) @lru_cache(maxsize=128)
def get_value(cls, content_type: Optional[str], text: Optional[str] = 'test') -> str: def get_value(cls, content_type: Optional[str], text: Optional[str] = 'test') -> str:
"""Determine the appropriate character encoding from a content-type header. """Determine the appropriate character encoding from a content-type header.
+2 -2
View File
@@ -12,7 +12,7 @@ from scrapling.core._types import Dict, Union
from scrapling.core.utils import lru_cache from scrapling.core.utils import lru_cache
@lru_cache(128, typed=True) @lru_cache(10, typed=True)
def generate_convincing_referer(url: str) -> str: def generate_convincing_referer(url: str) -> str:
"""Takes the domain from the URL without the subdomain/suffix and make it look like you were searching google for this website """Takes the domain from the URL without the subdomain/suffix and make it look like you were searching google for this website
@@ -26,7 +26,7 @@ def generate_convincing_referer(url: str) -> str:
return f'https://www.google.com/search?q={website_name}' return f'https://www.google.com/search?q={website_name}'
@lru_cache(128, typed=True) @lru_cache(1, typed=True)
def get_os_name() -> Union[str, None]: def get_os_name() -> Union[str, None]:
"""Get the current OS name in the same format needed for browserforge """Get the current OS name in the same format needed for browserforge
+1 -1
View File
@@ -110,7 +110,7 @@ def construct_cdp_url(cdp_url: str, query_params: Optional[Dict] = None) -> str:
raise ValueError(f"Invalid CDP URL: {str(e)}") raise ValueError(f"Invalid CDP URL: {str(e)}")
@lru_cache(126, typed=True) @lru_cache(10, typed=True)
def js_bypass_path(filename: str) -> str: def js_bypass_path(filename: str) -> str:
"""Takes the base filename of JS file inside the `bypasses` folder then return the full path of it """Takes the base filename of JS file inside the `bypasses` folder then return the full path of it