test: Rewrite parser tests to a cleaner version and adding more tests
This commit is contained in:
+90
-122
@@ -1,129 +1,97 @@
|
||||
import unittest
|
||||
import pytest
|
||||
|
||||
from scrapling.engines.toolbelt.custom import ResponseEncoding, StatusText
|
||||
|
||||
|
||||
class TestPlayWrightFetcher(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.content_type_map = {
|
||||
# A map generated by ChatGPT for most possible `content_type` values and the expected outcome
|
||||
'text/html; charset=UTF-8': 'UTF-8',
|
||||
'text/html; charset=ISO-8859-1': 'ISO-8859-1',
|
||||
'text/html': 'ISO-8859-1',
|
||||
'application/json; charset=UTF-8': 'UTF-8',
|
||||
'application/json': 'utf-8',
|
||||
'text/json': 'utf-8',
|
||||
'application/javascript; charset=UTF-8': 'UTF-8',
|
||||
'application/javascript': 'utf-8',
|
||||
'text/plain; charset=UTF-8': 'UTF-8',
|
||||
'text/plain; charset=ISO-8859-1': 'ISO-8859-1',
|
||||
'text/plain': 'ISO-8859-1',
|
||||
'application/xhtml+xml; charset=UTF-8': 'UTF-8',
|
||||
'application/xhtml+xml': 'utf-8',
|
||||
'text/html; charset=windows-1252': 'windows-1252',
|
||||
'application/json; charset=windows-1252': 'windows-1252',
|
||||
'text/plain; charset=windows-1252': 'windows-1252',
|
||||
'text/html; charset="UTF-8"': 'UTF-8',
|
||||
'text/html; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'text/html; charset="windows-1252"': 'windows-1252',
|
||||
'application/json; charset="UTF-8"': 'UTF-8',
|
||||
'application/json; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'application/json; charset="windows-1252"': 'windows-1252',
|
||||
'text/json; charset="UTF-8"': 'UTF-8',
|
||||
'application/javascript; charset="UTF-8"': 'UTF-8',
|
||||
'application/javascript; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'text/plain; charset="UTF-8"': 'UTF-8',
|
||||
'text/plain; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'text/plain; charset="windows-1252"': 'windows-1252',
|
||||
'application/xhtml+xml; charset="UTF-8"': 'UTF-8',
|
||||
'application/xhtml+xml; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'application/xhtml+xml; charset="windows-1252"': 'windows-1252',
|
||||
'text/html; charset="US-ASCII"': 'US-ASCII',
|
||||
'application/json; charset="US-ASCII"': 'US-ASCII',
|
||||
'text/plain; charset="US-ASCII"': 'US-ASCII',
|
||||
'text/html; charset="Shift_JIS"': 'Shift_JIS',
|
||||
'application/json; charset="Shift_JIS"': 'Shift_JIS',
|
||||
'text/plain; charset="Shift_JIS"': 'Shift_JIS',
|
||||
'application/xml; charset="UTF-8"': 'UTF-8',
|
||||
'application/xml; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'application/xml': 'utf-8',
|
||||
'text/xml; charset="UTF-8"': 'UTF-8',
|
||||
'text/xml; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'text/xml': 'utf-8'
|
||||
}
|
||||
self.status_map = {
|
||||
100: "Continue",
|
||||
101: "Switching Protocols",
|
||||
102: "Processing",
|
||||
103: "Early Hints",
|
||||
200: "OK",
|
||||
201: "Created",
|
||||
202: "Accepted",
|
||||
203: "Non-Authoritative Information",
|
||||
204: "No Content",
|
||||
205: "Reset Content",
|
||||
206: "Partial Content",
|
||||
207: "Multi-Status",
|
||||
208: "Already Reported",
|
||||
226: "IM Used",
|
||||
300: "Multiple Choices",
|
||||
301: "Moved Permanently",
|
||||
302: "Found",
|
||||
303: "See Other",
|
||||
304: "Not Modified",
|
||||
305: "Use Proxy",
|
||||
307: "Temporary Redirect",
|
||||
308: "Permanent Redirect",
|
||||
400: "Bad Request",
|
||||
401: "Unauthorized",
|
||||
402: "Payment Required",
|
||||
403: "Forbidden",
|
||||
404: "Not Found",
|
||||
405: "Method Not Allowed",
|
||||
406: "Not Acceptable",
|
||||
407: "Proxy Authentication Required",
|
||||
408: "Request Timeout",
|
||||
409: "Conflict",
|
||||
410: "Gone",
|
||||
411: "Length Required",
|
||||
412: "Precondition Failed",
|
||||
413: "Payload Too Large",
|
||||
414: "URI Too Long",
|
||||
415: "Unsupported Media Type",
|
||||
416: "Range Not Satisfiable",
|
||||
417: "Expectation Failed",
|
||||
418: "I'm a teapot",
|
||||
421: "Misdirected Request",
|
||||
422: "Unprocessable Entity",
|
||||
423: "Locked",
|
||||
424: "Failed Dependency",
|
||||
425: "Too Early",
|
||||
426: "Upgrade Required",
|
||||
428: "Precondition Required",
|
||||
429: "Too Many Requests",
|
||||
431: "Request Header Fields Too Large",
|
||||
451: "Unavailable For Legal Reasons",
|
||||
500: "Internal Server Error",
|
||||
501: "Not Implemented",
|
||||
502: "Bad Gateway",
|
||||
503: "Service Unavailable",
|
||||
504: "Gateway Timeout",
|
||||
505: "HTTP Version Not Supported",
|
||||
506: "Variant Also Negotiates",
|
||||
507: "Insufficient Storage",
|
||||
508: "Loop Detected",
|
||||
510: "Not Extended",
|
||||
511: "Network Authentication Required"
|
||||
}
|
||||
@pytest.fixture
|
||||
def content_type_map():
|
||||
return {
|
||||
# A map generated by ChatGPT for most possible `content_type` values and the expected outcome
|
||||
'text/html; charset=UTF-8': 'UTF-8',
|
||||
'text/html; charset=ISO-8859-1': 'ISO-8859-1',
|
||||
'text/html': 'ISO-8859-1',
|
||||
'application/json; charset=UTF-8': 'UTF-8',
|
||||
'application/json': 'utf-8',
|
||||
'text/json': 'utf-8',
|
||||
'application/javascript; charset=UTF-8': 'UTF-8',
|
||||
'application/javascript': 'utf-8',
|
||||
'text/plain; charset=UTF-8': 'UTF-8',
|
||||
'text/plain; charset=ISO-8859-1': 'ISO-8859-1',
|
||||
'text/plain': 'ISO-8859-1',
|
||||
'application/xhtml+xml; charset=UTF-8': 'UTF-8',
|
||||
'application/xhtml+xml': 'utf-8',
|
||||
'text/html; charset=windows-1252': 'windows-1252',
|
||||
'application/json; charset=windows-1252': 'windows-1252',
|
||||
'text/plain; charset=windows-1252': 'windows-1252',
|
||||
'text/html; charset="UTF-8"': 'UTF-8',
|
||||
'text/html; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'text/html; charset="windows-1252"': 'windows-1252',
|
||||
'application/json; charset="UTF-8"': 'UTF-8',
|
||||
'application/json; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'application/json; charset="windows-1252"': 'windows-1252',
|
||||
'text/json; charset="UTF-8"': 'UTF-8',
|
||||
'application/javascript; charset="UTF-8"': 'UTF-8',
|
||||
'application/javascript; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'text/plain; charset="UTF-8"': 'UTF-8',
|
||||
'text/plain; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'text/plain; charset="windows-1252"': 'windows-1252',
|
||||
'application/xhtml+xml; charset="UTF-8"': 'UTF-8',
|
||||
'application/xhtml+xml; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'application/xhtml+xml; charset="windows-1252"': 'windows-1252',
|
||||
'text/html; charset="US-ASCII"': 'US-ASCII',
|
||||
'application/json; charset="US-ASCII"': 'US-ASCII',
|
||||
'text/plain; charset="US-ASCII"': 'US-ASCII',
|
||||
'text/html; charset="Shift_JIS"': 'Shift_JIS',
|
||||
'application/json; charset="Shift_JIS"': 'Shift_JIS',
|
||||
'text/plain; charset="Shift_JIS"': 'Shift_JIS',
|
||||
'application/xml; charset="UTF-8"': 'UTF-8',
|
||||
'application/xml; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'application/xml': 'utf-8',
|
||||
'text/xml; charset="UTF-8"': 'UTF-8',
|
||||
'text/xml; charset="ISO-8859-1"': 'ISO-8859-1',
|
||||
'text/xml': 'utf-8'
|
||||
}
|
||||
|
||||
def test_parsing_content_type(self):
|
||||
"""Test if parsing different types of content-type returns the expected result"""
|
||||
for header_value, expected_encoding in self.content_type_map.items():
|
||||
self.assertEqual(ResponseEncoding.get_value(header_value), expected_encoding)
|
||||
|
||||
def test_parsing_response_status(self):
|
||||
"""Test if using different http responses' status codes returns the expected result"""
|
||||
for status_code, expected_status_text in self.status_map.items():
|
||||
self.assertEqual(StatusText.get(status_code), expected_status_text)
|
||||
@pytest.fixture
|
||||
def status_map():
|
||||
return {
|
||||
100: "Continue", 101: "Switching Protocols", 102: "Processing", 103: "Early Hints",
|
||||
200: "OK", 201: "Created", 202: "Accepted", 203: "Non-Authoritative Information",
|
||||
204: "No Content", 205: "Reset Content", 206: "Partial Content", 207: "Multi-Status",
|
||||
208: "Already Reported", 226: "IM Used", 300: "Multiple Choices",
|
||||
301: "Moved Permanently", 302: "Found", 303: "See Other", 304: "Not Modified",
|
||||
305: "Use Proxy", 307: "Temporary Redirect", 308: "Permanent Redirect",
|
||||
400: "Bad Request", 401: "Unauthorized", 402: "Payment Required", 403: "Forbidden",
|
||||
404: "Not Found", 405: "Method Not Allowed", 406: "Not Acceptable",
|
||||
407: "Proxy Authentication Required", 408: "Request Timeout", 409: "Conflict",
|
||||
410: "Gone", 411: "Length Required", 412: "Precondition Failed",
|
||||
413: "Payload Too Large", 414: "URI Too Long", 415: "Unsupported Media Type",
|
||||
416: "Range Not Satisfiable", 417: "Expectation Failed", 418: "I'm a teapot",
|
||||
421: "Misdirected Request", 422: "Unprocessable Entity", 423: "Locked",
|
||||
424: "Failed Dependency", 425: "Too Early", 426: "Upgrade Required",
|
||||
428: "Precondition Required", 429: "Too Many Requests",
|
||||
431: "Request Header Fields Too Large", 451: "Unavailable For Legal Reasons",
|
||||
500: "Internal Server Error", 501: "Not Implemented", 502: "Bad Gateway",
|
||||
503: "Service Unavailable", 504: "Gateway Timeout",
|
||||
505: "HTTP Version Not Supported", 506: "Variant Also Negotiates",
|
||||
507: "Insufficient Storage", 508: "Loop Detected", 510: "Not Extended",
|
||||
511: "Network Authentication Required"
|
||||
}
|
||||
|
||||
self.assertEqual(StatusText.get(1000), "Unknown Status Code")
|
||||
|
||||
def test_parsing_content_type(content_type_map):
|
||||
"""Test if parsing different types of content-type returns the expected result"""
|
||||
for header_value, expected_encoding in content_type_map.items():
|
||||
assert ResponseEncoding.get_value(header_value) == expected_encoding
|
||||
|
||||
|
||||
def test_parsing_response_status(status_map):
|
||||
"""Test if using different http responses' status codes returns the expected result"""
|
||||
for status_code, expected_status_text in status_map.items():
|
||||
assert StatusText.get(status_code) == expected_status_text
|
||||
|
||||
|
||||
def test_unknown_status_code():
|
||||
"""Test handling of an unknown status code"""
|
||||
assert StatusText.get(1000) == "Unknown Status Code"
|
||||
|
||||
+256
-214
@@ -1,288 +1,330 @@
|
||||
|
||||
import pickle
|
||||
import unittest
|
||||
import time
|
||||
|
||||
import pytest
|
||||
from cssselect import SelectorError, SelectorSyntaxError
|
||||
|
||||
from scrapling import Adaptor
|
||||
|
||||
|
||||
class TestParser(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.html = '''
|
||||
<html>
|
||||
<head>
|
||||
<title>Complex Web Page</title>
|
||||
<style>
|
||||
.hidden { display: none; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<header>
|
||||
<nav>
|
||||
<ul>
|
||||
<li><a href="#home">Home</a></li>
|
||||
<li><a href="#about">About</a></li>
|
||||
<li><a href="#contact">Contact</a></li>
|
||||
</ul>
|
||||
</nav>
|
||||
</header>
|
||||
<main>
|
||||
<section id="products" schema='{"jsonable": "data"}'>
|
||||
<h2>Products</h2>
|
||||
<div class="product-list">
|
||||
<article class="product" data-id="1">
|
||||
<h3>Product 1</h3>
|
||||
<p class="description">This is product 1</p>
|
||||
<span class="price">$10.99</span>
|
||||
<div class="hidden stock">In stock: 5</div>
|
||||
</article>
|
||||
<article class="product" data-id="2">
|
||||
<h3>Product 2</h3>
|
||||
<p class="description">This is product 2</p>
|
||||
<span class="price">$20.99</span>
|
||||
<div class="hidden stock">In stock: 3</div>
|
||||
</article>
|
||||
<article class="product" data-id="3">
|
||||
<h3>Product 3</h3>
|
||||
<p class="description">This is product 3</p>
|
||||
<span class="price">$15.99</span>
|
||||
<div class="hidden stock">Out of stock</div>
|
||||
</article>
|
||||
@pytest.fixture
|
||||
def html_content():
|
||||
return '''
|
||||
<html>
|
||||
<head>
|
||||
<title>Complex Web Page</title>
|
||||
<style>
|
||||
.hidden { display: none; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<header>
|
||||
<nav>
|
||||
<ul>
|
||||
<li><a href="#home">Home</a></li>
|
||||
<li><a href="#about">About</a></li>
|
||||
<li><a href="#contact">Contact</a></li>
|
||||
</ul>
|
||||
</nav>
|
||||
</header>
|
||||
<main>
|
||||
<section id="products" schema='{"jsonable": "data"}'>
|
||||
<h2>Products</h2>
|
||||
<div class="product-list">
|
||||
<article class="product" data-id="1">
|
||||
<h3>Product 1</h3>
|
||||
<p class="description">This is product 1</p>
|
||||
<span class="price">$10.99</span>
|
||||
<div class="hidden stock">In stock: 5</div>
|
||||
</article>
|
||||
<article class="product" data-id="2">
|
||||
<h3>Product 2</h3>
|
||||
<p class="description">This is product 2</p>
|
||||
<span class="price">$20.99</span>
|
||||
<div class="hidden stock">In stock: 3</div>
|
||||
</article>
|
||||
<article class="product" data-id="3">
|
||||
<h3>Product 3</h3>
|
||||
<p class="description">This is product 3</p>
|
||||
<span class="price">$15.99</span>
|
||||
<div class="hidden stock">Out of stock</div>
|
||||
</article>
|
||||
</div>
|
||||
</section>
|
||||
<section id="reviews">
|
||||
<h2>Customer Reviews</h2>
|
||||
<div class="review-list">
|
||||
<div class="review" data-rating="5">
|
||||
<p class="review-text">Great product!</p>
|
||||
<span class="reviewer">John Doe</span>
|
||||
</div>
|
||||
</section>
|
||||
<section id="reviews">
|
||||
<h2>Customer Reviews</h2>
|
||||
<div class="review-list">
|
||||
<div class="review" data-rating="5">
|
||||
<p class="review-text">Great product!</p>
|
||||
<span class="reviewer">John Doe</span>
|
||||
</div>
|
||||
<div class="review" data-rating="4">
|
||||
<p class="review-text">Good value for money.</p>
|
||||
<span class="reviewer">Jane Smith</span>
|
||||
</div>
|
||||
<div class="review" data-rating="4">
|
||||
<p class="review-text">Good value for money.</p>
|
||||
<span class="reviewer">Jane Smith</span>
|
||||
</div>
|
||||
</section>
|
||||
</main>
|
||||
<footer>
|
||||
<p>© 2024 Our Company</p>
|
||||
</footer>
|
||||
<script id="page-data" type="application/json">
|
||||
{"lastUpdated": "2024-09-22T10:30:00Z", "totalProducts": 3}
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
'''
|
||||
self.page = Adaptor(self.html, auto_match=False)
|
||||
</div>
|
||||
</section>
|
||||
</main>
|
||||
<footer>
|
||||
<p>© 2024 Our Company</p>
|
||||
</footer>
|
||||
<script id="page-data" type="application/json">
|
||||
{"lastUpdated": "2024-09-22T10:30:00Z", "totalProducts": 3}
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
'''
|
||||
|
||||
def test_css_selector(self):
|
||||
"""Test Selecting elements with complex CSS selectors"""
|
||||
elements = self.page.css('main #products .product-list article.product')
|
||||
self.assertEqual(len(elements), 3)
|
||||
|
||||
in_stock_products = self.page.css(
|
||||
@pytest.fixture
|
||||
def page(html_content):
|
||||
return Adaptor(html_content, auto_match=False)
|
||||
|
||||
|
||||
# CSS Selector Tests
|
||||
class TestCSSSelectors:
|
||||
def test_basic_product_selection(self, page):
|
||||
"""Test selecting all product elements"""
|
||||
elements = page.css('main #products .product-list article.product')
|
||||
assert len(elements) == 3
|
||||
|
||||
def test_in_stock_product_selection(self, page):
|
||||
"""Test selecting in-stock products"""
|
||||
in_stock_products = page.css(
|
||||
'main #products .product-list article.product:not(:contains("Out of stock"))')
|
||||
self.assertEqual(len(in_stock_products), 2)
|
||||
assert len(in_stock_products) == 2
|
||||
|
||||
def test_xpath_selector(self):
|
||||
"""Test Selecting elements with Complex XPath selectors"""
|
||||
reviews = self.page.xpath(
|
||||
|
||||
# XPath Selector Tests
|
||||
class TestXPathSelectors:
|
||||
def test_high_rating_reviews(self, page):
|
||||
"""Test selecting reviews with high ratings"""
|
||||
reviews = page.xpath(
|
||||
'//section[@id="reviews"]//div[contains(@class, "review") and @data-rating >= 4]'
|
||||
)
|
||||
self.assertEqual(len(reviews), 2)
|
||||
assert len(reviews) == 2
|
||||
|
||||
high_priced_products = self.page.xpath(
|
||||
def test_high_priced_products(self, page):
|
||||
"""Test selecting products above a certain price"""
|
||||
high_priced_products = page.xpath(
|
||||
'//article[contains(@class, "product")]'
|
||||
'[number(translate(substring-after(.//span[@class="price"], "$"), ",", "")) > 15]'
|
||||
)
|
||||
self.assertEqual(len(high_priced_products), 2)
|
||||
assert len(high_priced_products) == 2
|
||||
|
||||
def test_find_by_text(self):
|
||||
"""Test Selecting elements with Text matching"""
|
||||
stock_info = self.page.find_by_regex(r'In stock: \d+', first_match=False)
|
||||
self.assertEqual(len(stock_info), 2)
|
||||
|
||||
stock_info = self.page.find_by_regex(r'In stock: \d+', first_match=True, case_sensitive=True)
|
||||
self.assertEqual(stock_info.text, 'In stock: 5')
|
||||
# Text Matching Tests
|
||||
class TestTextMatching:
|
||||
def test_regex_multiple_matches(self, page):
|
||||
"""Test finding multiple matches with regex"""
|
||||
stock_info = page.find_by_regex(r'In stock: \d+', first_match=False)
|
||||
assert len(stock_info) == 2
|
||||
|
||||
stock_info = self.page.find_by_text(r'In stock:', partial=True, first_match=False)
|
||||
self.assertEqual(len(stock_info), 2)
|
||||
def test_regex_first_match(self, page):
|
||||
"""Test finding the first match with regex"""
|
||||
stock_info = page.find_by_regex(r'In stock: \d+', first_match=True, case_sensitive=True)
|
||||
assert stock_info.text == 'In stock: 5'
|
||||
|
||||
out_of_stock = self.page.find_by_text('Out of stock', partial=False, first_match=False)
|
||||
self.assertEqual(len(out_of_stock), 1)
|
||||
def test_partial_text_match(self, page):
|
||||
"""Test finding elements with partial text match"""
|
||||
stock_info = page.find_by_text(r'In stock:', partial=True, first_match=False)
|
||||
assert len(stock_info) == 2
|
||||
|
||||
def test_find_similar_elements(self):
|
||||
"""Test Finding similar elements of an element"""
|
||||
first_product = self.page.css_first('.product')
|
||||
def test_exact_text_match(self, page):
|
||||
"""Test finding elements with exact text match"""
|
||||
out_of_stock = page.find_by_text('Out of stock', partial=False, first_match=False)
|
||||
assert len(out_of_stock) == 1
|
||||
|
||||
|
||||
# Similar Elements Tests
|
||||
class TestSimilarElements:
|
||||
def test_finding_similar_products(self, page):
|
||||
"""Test finding similar product elements"""
|
||||
first_product = page.css_first('.product')
|
||||
similar_products = first_product.find_similar()
|
||||
self.assertEqual(len(similar_products), 2)
|
||||
assert len(similar_products) == 2
|
||||
|
||||
first_review = self.page.find('div', class_='review')
|
||||
def test_finding_similar_reviews(self, page):
|
||||
"""Test finding similar review elements with additional filtering"""
|
||||
first_review = page.find('div', class_='review')
|
||||
similar_high_rated_reviews = [
|
||||
review
|
||||
for review in first_review.find_similar()
|
||||
if int(review.attrib.get('data-rating', 0)) >= 4
|
||||
]
|
||||
self.assertEqual(len(similar_high_rated_reviews), 1)
|
||||
assert len(similar_high_rated_reviews) == 1
|
||||
|
||||
def test_expected_errors(self):
|
||||
"""Test errors that should raised if it does"""
|
||||
with self.assertRaises(ValueError):
|
||||
|
||||
# Error Handling Tests
|
||||
class TestErrorHandling:
|
||||
def test_invalid_adaptor_initialization(self):
|
||||
"""Test various invalid Adaptor initializations"""
|
||||
# No arguments
|
||||
with pytest.raises(ValueError):
|
||||
_ = Adaptor(auto_match=False)
|
||||
|
||||
with self.assertRaises(TypeError):
|
||||
# Invalid argument types
|
||||
with pytest.raises(TypeError):
|
||||
_ = Adaptor(root="ayo", auto_match=False)
|
||||
|
||||
with self.assertRaises(TypeError):
|
||||
with pytest.raises(TypeError):
|
||||
_ = Adaptor(text=1, auto_match=False)
|
||||
|
||||
with self.assertRaises(TypeError):
|
||||
with pytest.raises(TypeError):
|
||||
_ = Adaptor(body=1, auto_match=False)
|
||||
|
||||
with self.assertRaises(ValueError):
|
||||
_ = Adaptor(self.html, storage=object, auto_match=True)
|
||||
def test_invalid_storage(self, page, html_content):
|
||||
"""Test invalid storage parameter"""
|
||||
with pytest.raises(ValueError):
|
||||
_ = Adaptor(html_content, storage=object, auto_match=True)
|
||||
|
||||
def test_pickleable(self):
|
||||
"""Test that objects aren't pickleable"""
|
||||
table = self.page.css('.product-list')[0]
|
||||
with self.assertRaises(TypeError): # Adaptors
|
||||
def test_bad_selectors(self, page):
|
||||
"""Test handling of invalid selectors"""
|
||||
with pytest.raises((SelectorError, SelectorSyntaxError)):
|
||||
page.css('4 ayo')
|
||||
|
||||
with pytest.raises((SelectorError, SelectorSyntaxError)):
|
||||
page.xpath('4 ayo')
|
||||
|
||||
|
||||
# Pickling and Object Representation Tests
|
||||
class TestPicklingAndRepresentation:
|
||||
def test_unpickleable_objects(self, page):
|
||||
"""Test that Adaptor objects cannot be pickled"""
|
||||
table = page.css('.product-list')[0]
|
||||
with pytest.raises(TypeError):
|
||||
pickle.dumps(table)
|
||||
|
||||
with self.assertRaises(TypeError): # Adaptor
|
||||
with pytest.raises(TypeError):
|
||||
pickle.dumps(table[0])
|
||||
|
||||
def test_overridden(self):
|
||||
"""Test overridden functions"""
|
||||
table = self.page.css('.product-list')[0]
|
||||
self.assertTrue(issubclass(type(table.__str__()), str))
|
||||
self.assertTrue(issubclass(type(table.__repr__()), str))
|
||||
self.assertTrue(issubclass(type(table.attrib.__str__()), str))
|
||||
self.assertTrue(issubclass(type(table.attrib.__repr__()), str))
|
||||
def test_string_representations(self, page):
|
||||
"""Test custom string representations of objects"""
|
||||
table = page.css('.product-list')[0]
|
||||
assert issubclass(type(table.__str__()), str)
|
||||
assert issubclass(type(table.__repr__()), str)
|
||||
assert issubclass(type(table.attrib.__str__()), str)
|
||||
assert issubclass(type(table.attrib.__repr__()), str)
|
||||
|
||||
def test_bad_selector(self):
|
||||
"""Test object can handle bad selector"""
|
||||
with self.assertRaises((SelectorError, SelectorSyntaxError,)):
|
||||
self.page.css('4 ayo')
|
||||
|
||||
with self.assertRaises((SelectorError, SelectorSyntaxError,)):
|
||||
self.page.xpath('4 ayo')
|
||||
|
||||
def test_selectors_generation(self):
|
||||
"""Try to create selectors for all elements in the page"""
|
||||
def _traverse(element: Adaptor):
|
||||
self.assertTrue(type(element.generate_css_selector) is str)
|
||||
self.assertTrue(type(element.generate_xpath_selector) is str)
|
||||
for branch in element.children:
|
||||
_traverse(branch)
|
||||
|
||||
_traverse(self.page)
|
||||
|
||||
def test_getting_all_text(self):
|
||||
"""Test getting all text"""
|
||||
self.assertNotEqual(self.page.get_all_text(), '')
|
||||
|
||||
def test_element_navigation(self):
|
||||
"""Test moving in the page from selected element"""
|
||||
table = self.page.css('.product-list')[0]
|
||||
|
||||
self.assertIsNot(table.path, [])
|
||||
self.assertNotEqual(table.html_content, '')
|
||||
self.assertNotEqual(table.prettify(), '')
|
||||
# Navigation and Traversal Tests
|
||||
class TestElementNavigation:
|
||||
def test_basic_navigation_properties(self, page):
|
||||
"""Test basic navigation properties of elements"""
|
||||
table = page.css('.product-list')[0]
|
||||
assert table.path is not None
|
||||
assert table.html_content != ''
|
||||
assert table.prettify() != ''
|
||||
|
||||
def test_parent_and_sibling_navigation(self, page):
|
||||
"""Test parent and sibling navigation"""
|
||||
table = page.css('.product-list')[0]
|
||||
parent = table.parent
|
||||
self.assertEqual(parent.attrib['id'], 'products')
|
||||
|
||||
children = table.children
|
||||
self.assertEqual(len(children), 3)
|
||||
assert parent.attrib['id'] == 'products'
|
||||
|
||||
parent_siblings = parent.siblings
|
||||
self.assertEqual(len(parent_siblings), 1)
|
||||
assert len(parent_siblings) == 1
|
||||
|
||||
child = table.find({'data-id': "1"})
|
||||
def test_child_navigation(self, page):
|
||||
"""Test child navigation"""
|
||||
table = page.css('.product-list')[0]
|
||||
children = table.children
|
||||
assert len(children) == 3
|
||||
|
||||
def test_next_and_previous_navigation(self, page):
|
||||
"""Test next and previous element navigation"""
|
||||
child = page.css('.product-list')[0].find({'data-id': "1"})
|
||||
next_element = child.next
|
||||
self.assertEqual(next_element.attrib['data-id'], '2')
|
||||
assert next_element.attrib['data-id'] == '2'
|
||||
|
||||
prev_element = next_element.previous
|
||||
self.assertEqual(prev_element.tag, child.tag)
|
||||
assert prev_element.tag == child.tag
|
||||
|
||||
all_prices = self.page.css('.price')
|
||||
def test_ancestor_finding(self, page):
|
||||
"""Test finding ancestors of elements"""
|
||||
all_prices = page.css('.price')
|
||||
products_with_prices = [
|
||||
price.find_ancestor(lambda p: p.has_class('product'))
|
||||
for price in all_prices
|
||||
]
|
||||
self.assertEqual(len(products_with_prices), 3)
|
||||
assert len(products_with_prices) == 3
|
||||
|
||||
def test_empty_return(self):
|
||||
"""Test cases where functions shouldn't have results"""
|
||||
test_html = """
|
||||
<html>
|
||||
<span id="a"><a></a><!--comment--></span>
|
||||
<span id="b"><!--comment--><a></a></span>
|
||||
</html>"""
|
||||
soup = Adaptor(test_html, auto_match=False, keep_comments=False)
|
||||
html_tag = soup.css('html')[0]
|
||||
self.assertEqual(html_tag.path, [])
|
||||
self.assertEqual(html_tag.siblings, [])
|
||||
self.assertEqual(html_tag.parent, None)
|
||||
self.assertEqual(html_tag.find_ancestor(lambda e: e), None)
|
||||
|
||||
self.assertEqual(soup.css('#a a')[0].next, None)
|
||||
self.assertEqual(soup.css('#b a')[0].previous, None)
|
||||
|
||||
def test_text_to_json(self):
|
||||
"""Test converting text to json"""
|
||||
script_content = self.page.css('#page-data::text')[0]
|
||||
self.assertTrue(issubclass(type(script_content.sort()), str))
|
||||
# JSON and Attribute Tests
|
||||
class TestJSONAndAttributes:
|
||||
def test_json_conversion(self, page):
|
||||
"""Test converting content to JSON"""
|
||||
script_content = page.css('#page-data::text')[0]
|
||||
assert issubclass(type(script_content.sort()), str)
|
||||
page_data = script_content.json()
|
||||
self.assertEqual(page_data['totalProducts'], 3)
|
||||
self.assertTrue('lastUpdated' in page_data)
|
||||
assert page_data['totalProducts'] == 3
|
||||
assert 'lastUpdated' in page_data
|
||||
|
||||
def test_regex_on_text(self):
|
||||
"""Test doing regex on a selected text"""
|
||||
element = self.page.css('[data-id="1"] .price')[0]
|
||||
match = element.re_first(r'[\.\d]+')
|
||||
self.assertEqual(match, '10.99')
|
||||
match = element.text.re(r'(\d+)', replace_entities=False)
|
||||
self.assertEqual(len(match), 2)
|
||||
|
||||
def test_attribute_operations(self):
|
||||
"""Test operations on elements attributes"""
|
||||
products = self.page.css('.product')
|
||||
def test_attribute_operations(self, page):
|
||||
"""Test various attribute-related operations"""
|
||||
# Product ID extraction
|
||||
products = page.css('.product')
|
||||
product_ids = [product.attrib['data-id'] for product in products]
|
||||
self.assertEqual(product_ids, ['1', '2', '3'])
|
||||
self.assertTrue('data-id' in products[0].attrib)
|
||||
assert product_ids == ['1', '2', '3']
|
||||
assert 'data-id' in products[0].attrib
|
||||
|
||||
reviews = self.page.css('.review')
|
||||
# Review rating calculations
|
||||
reviews = page.css('.review')
|
||||
review_ratings = [int(review.attrib['data-rating']) for review in reviews]
|
||||
self.assertEqual(sum(review_ratings) / len(review_ratings), 4.5)
|
||||
assert sum(review_ratings) / len(review_ratings) == 4.5
|
||||
|
||||
# Attribute searching
|
||||
key_value = list(products[0].attrib.search_values('1', partial=False))
|
||||
self.assertEqual(list(key_value[0].keys()), ['data-id'])
|
||||
assert list(key_value[0].keys()) == ['data-id']
|
||||
|
||||
key_value = list(products[0].attrib.search_values('1', partial=True))
|
||||
self.assertEqual(list(key_value[0].keys()), ['data-id'])
|
||||
assert list(key_value[0].keys()) == ['data-id']
|
||||
|
||||
attr_json = self.page.css_first('#products').attrib['schema'].json()
|
||||
self.assertEqual(attr_json, {'jsonable': 'data'})
|
||||
self.assertEqual(type(self.page.css('#products')[0].attrib.json_string), bytes)
|
||||
|
||||
def test_performance(self):
|
||||
"""Test parsing and selecting speed"""
|
||||
import time
|
||||
large_html = '<html><body>' + '<div class="item">' * 5000 + '</div>' * 5000 + '</body></html>'
|
||||
|
||||
start_time = time.time()
|
||||
parsed = Adaptor(large_html, auto_match=False)
|
||||
elements = parsed.css('.item')
|
||||
end_time = time.time()
|
||||
|
||||
self.assertEqual(len(elements), 5000)
|
||||
# Converting 5000 elements to a class and doing operations on them will take time
|
||||
# Based on my tests with 100 runs, 1 loop each Scrapling (given the extra work/features) takes 10.4ms on average
|
||||
self.assertLess(end_time - start_time, 0.5) # Locally I test on 0.1 but on GitHub actions with browsers and threading sometimes closing adds fractions of seconds
|
||||
# JSON attribute conversion
|
||||
attr_json = page.css_first('#products').attrib['schema'].json()
|
||||
assert attr_json == {'jsonable': 'data'}
|
||||
assert isinstance(page.css('#products')[0].attrib.json_string, bytes)
|
||||
|
||||
|
||||
# Use `coverage run -m unittest --verbose tests/test_parser_functions.py` instead for the coverage report
|
||||
# if __name__ == '__main__':
|
||||
# unittest.main(verbosity=2)
|
||||
# Performance Test
|
||||
def test_large_html_parsing_performance():
|
||||
"""Test parsing and selecting performance on large HTML"""
|
||||
large_html = '<html><body>' + '<div class="item">' * 5000 + '</div>' * 5000 + '</body></html>'
|
||||
|
||||
start_time = time.time()
|
||||
parsed = Adaptor(large_html, auto_match=False)
|
||||
elements = parsed.css('.item')
|
||||
end_time = time.time()
|
||||
|
||||
assert len(elements) == 5000
|
||||
# Converting 5000 elements to a class and doing operations on them will take time
|
||||
# Based on my tests with 100 runs, 1 loop each Scrapling (given the extra work/features) takes 10.4ms on average
|
||||
assert end_time - start_time < 0.5 # Locally I test on 0.1 but on GitHub actions with browsers and threading sometimes closing adds fractions of seconds
|
||||
|
||||
|
||||
# Selector Generation Test
|
||||
def test_selectors_generation(page):
|
||||
"""Try to create selectors for all elements in the page"""
|
||||
|
||||
def _traverse(element: Adaptor):
|
||||
assert isinstance(element.generate_css_selector, str)
|
||||
assert isinstance(element.generate_xpath_selector, str)
|
||||
for branch in element.children:
|
||||
_traverse(branch)
|
||||
|
||||
_traverse(page)
|
||||
|
||||
|
||||
# Miscellaneous Tests
|
||||
def test_getting_all_text(page):
|
||||
"""Test getting all text from the page"""
|
||||
assert page.get_all_text() != ''
|
||||
|
||||
|
||||
def test_regex_on_text(page):
|
||||
"""Test regex operations on text"""
|
||||
element = page.css('[data-id="1"] .price')[0]
|
||||
match = element.re_first(r'[\.\d]+')
|
||||
assert match == '10.99'
|
||||
match = element.text.re(r'(\d+)', replace_entities=False)
|
||||
assert len(match) == 2
|
||||
|
||||
Reference in New Issue
Block a user