This commit is contained in:
Karim shoair
2025-11-26 19:49:23 +02:00
committed by GitHub
31 changed files with 1184 additions and 1849 deletions
+1 -1
View File
@@ -116,7 +116,7 @@ jobs:
with: with:
path: .tox path: .tox
# Include python version and os in the cache key # Include python version and os in the cache key
key: tox-v1-${{ runner.os }}-py${{ matrix.python-version }}-${{ hashFiles('tox.ini', 'pyproject.toml') }} key: tox-v1-${{ runner.os }}-py${{ matrix.python-version }}-${{ hashFiles('/Users/runner/work/Scrapling/pyproject.toml') }}
restore-keys: | restore-keys: |
tox-v1-${{ runner.os }}-py${{ matrix.python-version }}- tox-v1-${{ runner.os }}-py${{ matrix.python-version }}-
tox-v1-${{ runner.os }}- tox-v1-${{ runner.os }}-
+11 -10
View File
@@ -244,14 +244,15 @@ Scrapling isn't just powerful—it's also blazing fast, and the updates since ve
| # | Library | Time (ms) | vs Scrapling | | # | Library | Time (ms) | vs Scrapling |
|---|:-----------------:|:---------:|:------------:| |---|:-----------------:|:---------:|:------------:|
| 1 | Scrapling | 1.92 | 1.0x | | 1 | Scrapling | 1.99 | 1.0x |
| 2 | Parsel/Scrapy | 1.99 | 1.036x | | 2 | Parsel/Scrapy | 2.01 | 1.01x |
| 3 | Raw Lxml | 2.33 | 1.214x | | 3 | Raw Lxml | 2.5 | 1.256x |
| 4 | PyQuery | 20.61 | ~11x | | 4 | PyQuery | 22.93 | ~11.5x |
| 5 | Selectolax | 80.65 | ~42x | | 5 | Selectolax | 80.57 | ~40.5x |
| 6 | BS4 with Lxml | 1283.21 | ~698x | | 6 | BS4 with Lxml | 1541.37 | ~774.6x |
| 7 | MechanicalSoup | 1304.57 | ~679x | | 7 | MechanicalSoup | 1547.35 | ~777.6x |
| 8 | BS4 with html5lib | 3331.96 | ~1735x | | 8 | BS4 with html5lib | 3410.58 | ~1713.9x |
### Element Similarity & Text Search Performance ### Element Similarity & Text Search Performance
@@ -259,8 +260,8 @@ Scrapling's adaptive element finding capabilities significantly outperform alter
| Library | Time (ms) | vs Scrapling | | Library | Time (ms) | vs Scrapling |
|-------------|:---------:|:------------:| |-------------|:---------:|:------------:|
| Scrapling | 1.87 | 1.0x | | Scrapling | 2.46 | 1.0x |
| AutoScraper | 10.24 | 5.476x | | AutoScraper | 13.3 | 5.407x |
> All benchmarks represent averages of 100+ runs. See [benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py) for methodology. > All benchmarks represent averages of 100+ runs. See [benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py) for methodology.
+14 -13
View File
@@ -233,24 +233,25 @@ Scrapling ليس قوياً فقط - إنه أيضاً سريع بشكل مذه
### اختبار سرعة استخراج النص (5000 عنصر متداخل) ### اختبار سرعة استخراج النص (5000 عنصر متداخل)
| # | المكتبة | الوقت (ms) | vs Scrapling | | # | المكتبة | الوقت (ms) | vs Scrapling |
|---|:-----------------:|:---------:|:------------:| |---|:-----------------:|:----------:|:------------:|
| 1 | Scrapling | 1.92 | 1.0x | | 1 | Scrapling | 1.99 | 1.0x |
| 2 | Parsel/Scrapy | 1.99 | 1.036x | | 2 | Parsel/Scrapy | 2.01 | 1.01x |
| 3 | Raw Lxml | 2.33 | 1.214x | | 3 | Raw Lxml | 2.5 | 1.256x |
| 4 | PyQuery | 20.61 | ~11x | | 4 | PyQuery | 22.93 | ~11.5x |
| 5 | Selectolax | 80.65 | ~42x | | 5 | Selectolax | 80.57 | ~40.5x |
| 6 | BS4 with Lxml | 1283.21 | ~698x | | 6 | BS4 with Lxml | 1541.37 | ~774.6x |
| 7 | MechanicalSoup | 1304.57 | ~679x | | 7 | MechanicalSoup | 1547.35 | ~777.6x |
| 8 | BS4 with html5lib | 3331.96 | ~1735x | | 8 | BS4 with html5lib | 3410.58 | ~1713.9x |
### أداء تشابه العناصر والبحث النصي ### أداء تشابه العناصر والبحث النصي
قدرات العثور على العناصر التكيفية لـ Scrapling تتفوق بشكل كبير على البدائل: قدرات العثور على العناصر التكيفية لـ Scrapling تتفوق بشكل كبير على البدائل:
| المكتبة | الوقت (ms) | vs Scrapling | | المكتبة | الوقت (ms) | vs Scrapling |
|-------------|:---------:|:------------:| |-------------|:----------:|:------------:|
| Scrapling | 1.87 | 1.0x | | Scrapling | 2.46 | 1.0x |
| AutoScraper | 10.24 | 5.476x | | AutoScraper | 13.3 | 5.407x |
> تمثل جميع المعايير متوسطات أكثر من 100 تشغيل. انظر [benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py) للمنهجية. > تمثل جميع المعايير متوسطات أكثر من 100 تشغيل. انظر [benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py) للمنهجية.
+15 -14
View File
@@ -232,25 +232,26 @@ Scrapling不仅功能强大——它还速度极快,自0.3版本以来的更
### 文本提取速度测试(5000个嵌套元素) ### 文本提取速度测试(5000个嵌套元素)
| # | | 时间(ms) | vs Scrapling | | # | | 时间(ms) | vs Scrapling |
|---|:--------------:|:--------:|:------------:| |---|:-----------------:|:-------:|:------------:|
| 1 | Scrapling | 1.92 | 1.0x | | 1 | Scrapling | 1.99 | 1.0x |
| 2 | Parsel/Scrapy | 1.99 | 1.036x | | 2 | Parsel/Scrapy | 2.01 | 1.01x |
| 3 | Raw Lxml | 2.33 | 1.214x | | 3 | Raw Lxml | 2.5 | 1.256x |
| 4 | PyQuery | 20.61 | ~11x | | 4 | PyQuery | 22.93 | ~11.5x |
| 5 | Selectolax | 80.65 | ~42x | | 5 | Selectolax | 80.57 | ~40.5x |
| 6 | BS4 with Lxml | 1283.21 | ~698x | | 6 | BS4 with Lxml | 1541.37 | ~774.6x |
| 7 | MechanicalSoup | 1304.57 | ~679x | | 7 | MechanicalSoup | 1547.35 | ~777.6x |
| 8 |BS4 with html5lib| 3331.96 | ~1735x | | 8 | BS4 with html5lib | 3410.58 | ~1713.9x |
### 元素相似性和文本搜索性能 ### 元素相似性和文本搜索性能
Scrapling的自适应元素查找功能明显优于替代方案: Scrapling的自适应元素查找功能明显优于替代方案:
| | 时间(ms) | vs Scrapling | | | 时间(ms) | vs Scrapling |
|-------------|:--------:|:------------:| |-------------|:------:|:------------:|
| Scrapling | 1.87 | 1.0x | | Scrapling | 2.46 | 1.0x |
| AutoScraper | 10.24 | 5.476x | | AutoScraper | 13.3 | 5.407x |
> 所有基准测试代表100+次运行的平均值。请参阅[benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py)了解方法。 > 所有基准测试代表100+次运行的平均值。请参阅[benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py)了解方法。
+14 -13
View File
@@ -232,25 +232,26 @@ Scrapling ist nicht nur leistungsstark es ist auch blitzschnell, und die Upd
### Textextraktions-Geschwindigkeitstest (5000 verschachtelte Elemente) ### Textextraktions-Geschwindigkeitstest (5000 verschachtelte Elemente)
| # | Bibliothek | Zeit (ms) | vs Scrapling | | # | Bibliothek | Zeit (ms) | vs Scrapling |
|---|:--------------------:|:---------:|:------------:| |---|:-----------------:|:---------:|:------------:|
| 1 | Scrapling | 1.92 | 1.0x | | 1 | Scrapling | 1.99 | 1.0x |
| 2 | Parsel/Scrapy | 1.99 | 1.036x | | 2 | Parsel/Scrapy | 2.01 | 1.01x |
| 3 | Raw Lxml | 2.33 | 1.214x | | 3 | Raw Lxml | 2.5 | 1.256x |
| 4 | PyQuery | 20.61 | ~11x | | 4 | PyQuery | 22.93 | ~11.5x |
| 5 | Selectolax | 80.65 | ~42x | | 5 | Selectolax | 80.57 | ~40.5x |
| 6 | BS4 mit Lxml | 1283.21 | ~698x | | 6 | BS4 with Lxml | 1541.37 | ~774.6x |
| 7 | MechanicalSoup | 1304.57 | ~679x | | 7 | MechanicalSoup | 1547.35 | ~777.6x |
| 8 | BS4 mit html5lib | 3331.96 | ~1735x | | 8 | BS4 with html5lib | 3410.58 | ~1713.9x |
### Element-Ähnlichkeit & Textsuche-Leistung ### Element-Ähnlichkeit & Textsuche-Leistung
Scraplings adaptive Element-Finding-Fähigkeiten übertreffen Alternativen deutlich: Scraplings adaptive Element-Finding-Fähigkeiten übertreffen Alternativen deutlich:
| Bibliothek | Zeit (ms) | vs Scrapling | | Bibliothek | Zeit (ms) | vs Scrapling |
|-------------|:---------:|:------------:| |-------------|:---------:|:------------:|
| Scrapling | 1.87 | 1.0x | | Scrapling | 2.46 | 1.0x |
| AutoScraper | 10.24 | 5.476x | | AutoScraper | 13.3 | 5.407x |
> Alle Benchmarks stellen Durchschnittswerte von über 100 Durchläufen dar. Siehe [benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py) für die Methodik. > Alle Benchmarks stellen Durchschnittswerte von über 100 Durchläufen dar. Siehe [benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py) für die Methodik.
+15 -14
View File
@@ -232,25 +232,26 @@ Scrapling no solo es poderoso, también es increíblemente rápido, y las actual
### Prueba de Velocidad de Extracción de Texto (5000 elementos anidados) ### Prueba de Velocidad de Extracción de Texto (5000 elementos anidados)
| # | Biblioteca | Tiempo (ms) | vs Scrapling | | # | Biblioteca | Tiempo (ms) | vs Scrapling |
|---|:--------------------:|:-----------:|:------------:| |---|:-----------------:|:-----------:|:------------:|
| 1 | Scrapling | 1.92 | 1.0x | | 1 | Scrapling | 1.99 | 1.0x |
| 2 | Parsel/Scrapy | 1.99 | 1.036x | | 2 | Parsel/Scrapy | 2.01 | 1.01x |
| 3 | Raw Lxml | 2.33 | 1.214x | | 3 | Raw Lxml | 2.5 | 1.256x |
| 4 | PyQuery | 20.61 | ~11x | | 4 | PyQuery | 22.93 | ~11.5x |
| 5 | Selectolax | 80.65 | ~42x | | 5 | Selectolax | 80.57 | ~40.5x |
| 6 | BS4 con Lxml | 1283.21 | ~698x | | 6 | BS4 with Lxml | 1541.37 | ~774.6x |
| 7 | MechanicalSoup | 1304.57 | ~679x | | 7 | MechanicalSoup | 1547.35 | ~777.6x |
| 8 | BS4 con html5lib | 3331.96 | ~1735x | | 8 | BS4 with html5lib | 3410.58 | ~1713.9x |
### Rendimiento de Similitud de Elementos y Búsqueda de Texto ### Rendimiento de Similitud de Elementos y Búsqueda de Texto
Las capacidades de búsqueda adaptativa de elementos de Scrapling superan significativamente a las alternativas: Las capacidades de búsqueda adaptativa de elementos de Scrapling superan significativamente a las alternativas:
| Biblioteca | Tiempo (ms) | vs Scrapling | | Biblioteca | Tiempo (ms) | vs Scrapling |
|--------------|:-----------:|:------------:| |-------------|:-----------:|:------------:|
| Scrapling | 1.87 | 1.0x | | Scrapling | 2.46 | 1.0x |
| AutoScraper | 10.24 | 5.476x | | AutoScraper | 13.3 | 5.407x |
> Todos los benchmarks representan promedios de más de 100 ejecuciones. Ver [benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py) para la metodología. > Todos los benchmarks representan promedios de más de 100 ejecuciones. Ver [benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py) para la metodología.
+15 -14
View File
@@ -232,25 +232,26 @@ Scraplingは強力であるだけでなく、驚くほど高速で、バージ
### テキスト抽出速度テスト(5000個のネストされた要素) ### テキスト抽出速度テスト(5000個のネストされた要素)
| # | ライブラリ | 時間(ms) | vs Scrapling | | # | ライブラリ | 時間(ms) | vs Scrapling |
|---|:-------------------:|:--------:|:------------:| |---|:-----------------:|:-------:|:------------:|
| 1 | Scrapling | 1.92 | 1.0x | | 1 | Scrapling | 1.99 | 1.0x |
| 2 | Parsel/Scrapy | 1.99 | 1.036x | | 2 | Parsel/Scrapy | 2.01 | 1.01x |
| 3 | Raw Lxml | 2.33 | 1.214x | | 3 | Raw Lxml | 2.5 | 1.256x |
| 4 | PyQuery | 20.61 | ~11x | | 4 | PyQuery | 22.93 | ~11.5x |
| 5 | Selectolax | 80.65 | ~42x | | 5 | Selectolax | 80.57 | ~40.5x |
| 6 | BS4 with Lxml | 1283.21 | ~698x | | 6 | BS4 with Lxml | 1541.37 | ~774.6x |
| 7 | MechanicalSoup | 1304.57 | ~679x | | 7 | MechanicalSoup | 1547.35 | ~777.6x |
| 8 | BS4 with html5lib | 3331.96 | ~1735x | | 8 | BS4 with html5lib | 3410.58 | ~1713.9x |
### 要素類似性とテキスト検索のパフォーマンス ### 要素類似性とテキスト検索のパフォーマンス
Scraplingの適応型要素検索機能は代替手段を大幅に上回ります: Scraplingの適応型要素検索機能は代替手段を大幅に上回ります:
| ライブラリ | 時間(ms) | vs Scrapling | | ライブラリ | 時間(ms) | vs Scrapling |
|-------------|:--------:|:------------:| |-------------|:------:|:------------:|
| Scrapling | 1.87 | 1.0x | | Scrapling | 2.46 | 1.0x |
| AutoScraper | 10.24 | 5.476x | | AutoScraper | 13.3 | 5.407x |
> すべてのベンチマークは100回以上の実行の平均を表します。方法論については[benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py)を参照してください。 > すべてのベンチマークは100回以上の実行の平均を表します。方法論については[benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py)を参照してください。
+14 -13
View File
@@ -232,25 +232,26 @@ Scrapling не только мощный - он также невероятно
### Тест скорости извлечения текста (5000 вложенных элементов) ### Тест скорости извлечения текста (5000 вложенных элементов)
| # | Библиотека | Время (мс) | vs Scrapling | | # | Библиотека | Время (мс) | vs Scrapling |
|---|:--------------------:|:----------:|:------------:| |---|:-----------------:|:----------:|:------------:|
| 1 | Scrapling | 1.92 | 1.0x | | 1 | Scrapling | 1.99 | 1.0x |
| 2 | Parsel/Scrapy | 1.99 | 1.036x | | 2 | Parsel/Scrapy | 2.01 | 1.01x |
| 3 | Raw Lxml | 2.33 | 1.214x | | 3 | Raw Lxml | 2.5 | 1.256x |
| 4 | PyQuery | 20.61 | ~11x | | 4 | PyQuery | 22.93 | ~11.5x |
| 5 | Selectolax | 80.65 | ~42x | | 5 | Selectolax | 80.57 | ~40.5x |
| 6 | BS4 с Lxml | 1283.21 | ~698x | | 6 | BS4 with Lxml | 1541.37 | ~774.6x |
| 7 | MechanicalSoup | 1304.57 | ~679x | | 7 | MechanicalSoup | 1547.35 | ~777.6x |
| 8 | BS4 с html5lib | 3331.96 | ~1735x | | 8 | BS4 with html5lib | 3410.58 | ~1713.9x |
### Производительность подобия элементов и текстового поиска ### Производительность подобия элементов и текстового поиска
Возможности адаптивного поиска элементов Scrapling значительно превосходят альтернативы: Возможности адаптивного поиска элементов Scrapling значительно превосходят альтернативы:
| Библиотека | Время (мс) | vs Scrapling | | Библиотека | Время (мс) | vs Scrapling |
|-------------|:----------:|:------------:| |-------------|:----------:|:------------:|
| Scrapling | 1.87 | 1.0x | | Scrapling | 2.46 | 1.0x |
| AutoScraper | 10.24 | 5.476x | | AutoScraper | 13.3 | 5.407x |
> Все тесты производительности представляют собой средние значения более 100 запусков. См. [benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py) для методологии. > Все тесты производительности представляют собой средние значения более 100 запусков. См. [benchmarks.py](https://github.com/D4Vinci/Scrapling/blob/main/benchmarks.py) для методологии.
+11 -11
View File
@@ -8,20 +8,20 @@ Scrapling isn't just powerful—it's also blazing fast, and the updates since ve
| # | Library | Time (ms) | vs Scrapling | | # | Library | Time (ms) | vs Scrapling |
|---|:-----------------:|:---------:|:------------:| |---|:-----------------:|:---------:|:------------:|
| 1 | Scrapling | 1.92 | 1.0x | | 1 | Scrapling | 1.99 | 1.0x |
| 2 | Parsel/Scrapy | 1.99 | 1.036x | | 2 | Parsel/Scrapy | 2.01 | 1.01x |
| 3 | Raw Lxml | 2.33 | 1.214x | | 3 | Raw Lxml | 2.5 | 1.256x |
| 4 | PyQuery | 20.61 | ~11x | | 4 | PyQuery | 22.93 | ~11.5x |
| 5 | Selectolax | 80.65 | ~42x | | 5 | Selectolax | 80.57 | ~40.5x |
| 6 | BS4 with Lxml | 1283.21 | ~698x | | 6 | BS4 with Lxml | 1541.37 | ~774.6x |
| 7 | MechanicalSoup | 1304.57 | ~679x | | 7 | MechanicalSoup | 1547.35 | ~777.6x |
| 8 | BS4 with html5lib | 3331.96 | ~1735x | | 8 | BS4 with html5lib | 3410.58 | ~1713.9x |
### Element Similarity & Text Search Performance ### Element Similarity & Text Search Performance
Scrapling's adaptive element finding capabilities significantly outperform alternatives: Scrapling's adaptive element finding capabilities significantly outperform alternatives:
| Library | Time (ms) | vs Scrapling | | Library | Time (ms) | vs Scrapling |
|-------------|:---------:|:------------:| |-------------|:---------:|:------------:|
| Scrapling | 1.87 | 1.0x | | Scrapling | 2.46 | 1.0x |
| AutoScraper | 10.24 | 5.476x | | AutoScraper | 13.3 | 5.407x |
+1 -1
View File
@@ -57,7 +57,7 @@ The available configuration arguments are: `adaptive`, `huge_tree`, `keep_commen
### Set parser config per request ### Set parser config per request
As you probably understand, the logic above for setting the parser config will apply globally to all requests/fetches made through that class, and it's intended for simplicity. As you probably understand, the logic above for setting the parser config will apply globally to all requests/fetches made through that class, and it's intended for simplicity.
If your use case requires a different configuration for each request/fetch, you can pass a dictionary to the request method (`fetch`/`get`/`post`/...) to an argument named `custom_config`. If your use case requires a different configuration for each request/fetch, you can pass a dictionary to the request method (`fetch`/`get`/`post`/...) to an argument named `selector_config`.
## Response Object ## Response Object
The `Response` object is the same as the [Selector](../parsing/main_classes.md#selector) class, but it has additional details about the response, like response headers, status, cookies, etc., as shown below: The `Response` object is the same as the [Selector](../parsing/main_classes.md#selector) class, but it has additional details about the response, like response headers, status, cookies, etc., as shown below:
+5 -5
View File
@@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta"
[project] [project]
name = "scrapling" name = "scrapling"
# Static version instead of a dynamic version so we can get better layer caching while building docker, check the docker file to understand # Static version instead of a dynamic version so we can get better layer caching while building docker, check the docker file to understand
version = "0.3.9" version = "0.3.10"
description = "Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!" description = "Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!"
readme = {file = "docs/README.md", content-type = "text/markdown"} readme = {file = "docs/README.md", content-type = "text/markdown"}
license = {file = "LICENSE"} license = {file = "LICENSE"}
@@ -67,11 +67,11 @@ dependencies = [
fetchers = [ fetchers = [
"click>=8.3.0", "click>=8.3.0",
"curl_cffi>=0.13.0", "curl_cffi>=0.13.0",
"playwright>=1.55.0", "playwright>=1.56.0",
"patchright>=1.55.2", "patchright>=1.56.0",
"camoufox>=0.4.11", "camoufox>=0.4.11",
"geoip2>=5.1.0", "geoip2>=5.2.0",
"msgspec>=0.19.0", "msgspec>=0.20.0",
] ]
ai = [ ai = [
"mcp>=1.19.0", "mcp>=1.19.0",
+1 -1
View File
@@ -1,5 +1,5 @@
__author__ = "Karim Shoair (karim.shoair@pm.me)" __author__ = "Karim Shoair (karim.shoair@pm.me)"
__version__ = "0.3.9" __version__ = "0.3.10"
__copyright__ = "Copyright (c) 2024 Karim Shoair" __copyright__ = "Copyright (c) 2024 Karim Shoair"
from typing import Any, TYPE_CHECKING from typing import Any, TYPE_CHECKING
+95
View File
@@ -0,0 +1,95 @@
from scrapling.core._types import (
Dict,
Any,
List,
Tuple,
Optional,
)
# Parameter definitions for shell function signatures (defined once at module level)
# Mirrors TypedDict definitions from _types.py but runtime-accessible for IPython introspection
_REQUESTS_PARAMS = {
"params": Optional[Dict | List | Tuple],
"cookies": Any,
"auth": Optional[Tuple[str, str]],
"impersonate": Any,
"http3": Optional[bool],
"stealthy_headers": Optional[bool],
"proxies": Any,
"proxy": Optional[str],
"proxy_auth": Optional[Tuple[str, str]],
"timeout": Optional[int | float],
"headers": Any,
"retries": Optional[int],
"retry_delay": Optional[int],
"follow_redirects": Optional[bool],
"max_redirects": Optional[int],
"verify": Optional[bool],
"cert": Optional[str | Tuple[str, str]],
"selector_config": Optional[Dict],
}
_FETCH_PARAMS = {
"headless": bool,
"google_search": bool,
"hide_canvas": bool,
"disable_webgl": bool,
"real_chrome": bool,
"stealth": bool,
"wait": int | float,
"page_action": Optional[Any],
"proxy": Optional[str | Dict],
"locale": str,
"extra_headers": Optional[Dict[str, str]],
"useragent": Optional[str],
"cdp_url": Optional[str],
"timeout": int | float,
"disable_resources": bool,
"wait_selector": Optional[str],
"init_script": Optional[str],
"cookies": Optional[List[Dict]],
"network_idle": bool,
"load_dom": bool,
"wait_selector_state": Any,
"extra_flags": Optional[List[str]],
"additional_args": Optional[Dict],
"selector_config": Optional[Dict],
}
_STEALTHY_FETCH_PARAMS = {
"headless": bool,
"block_images": bool,
"disable_resources": bool,
"block_webrtc": bool,
"allow_webgl": bool,
"network_idle": bool,
"load_dom": bool,
"humanize": bool | float,
"solve_cloudflare": bool,
"wait": int | float,
"timeout": int | float,
"page_action": Optional[Any],
"wait_selector": Optional[str],
"init_script": Optional[str],
"addons": Optional[List[str]],
"wait_selector_state": Any,
"cookies": Optional[List[Dict]],
"google_search": bool,
"extra_headers": Optional[Dict[str, str]],
"proxy": Optional[str | Dict],
"os_randomize": bool,
"disable_ads": bool,
"geoip": bool,
"selector_config": Optional[Dict],
"additional_args": Optional[Dict],
}
# Mapping of function names to their parameter definitions
Signatures_map = {
"get": _REQUESTS_PARAMS,
"post": {**_REQUESTS_PARAMS, "data": Optional[Dict | str], "json": Optional[Dict | List]},
"put": {**_REQUESTS_PARAMS, "data": Optional[Dict | str], "json": Optional[Dict | List]},
"delete": _REQUESTS_PARAMS,
"fetch": _FETCH_PARAMS,
"stealthy_fetch": _STEALTHY_FETCH_PARAMS,
}
+14
View File
@@ -4,12 +4,15 @@ Type definitions for type checking purposes.
from typing import ( from typing import (
TYPE_CHECKING, TYPE_CHECKING,
TypedDict,
TypeAlias,
cast, cast,
overload, overload,
Any, Any,
Callable, Callable,
Dict, Dict,
Generator, Generator,
Generic,
Iterable, Iterable,
List, List,
Set, Set,
@@ -34,6 +37,17 @@ PageLoadStates = Literal["commit", "domcontentloaded", "load", "networkidle"]
extraction_types = Literal["text", "html", "markdown"] extraction_types = Literal["text", "html", "markdown"]
StrOrBytes = Union[str, bytes] StrOrBytes = Union[str, bytes]
if TYPE_CHECKING: # pragma: no cover
from typing_extensions import Unpack
else: # pragma: no cover
class _Unpack:
@staticmethod
def __getitem__(*args, **kwargs):
pass
Unpack = _Unpack()
try: try:
# Python 3.11+ # Python 3.11+
+49 -7
View File
@@ -1,13 +1,14 @@
# -*- coding: utf-8 -*- # -*- coding: utf-8 -*-
from re import sub as re_sub
from sys import stderr from sys import stderr
from functools import wraps from functools import wraps
from re import sub as re_sub
from collections import namedtuple from collections import namedtuple
from shlex import split as shlex_split from shlex import split as shlex_split
from inspect import signature, Parameter
from tempfile import mkstemp as make_temp_file from tempfile import mkstemp as make_temp_file
from urllib.parse import urlparse, urlunparse, parse_qsl
from argparse import ArgumentParser, SUPPRESS from argparse import ArgumentParser, SUPPRESS
from webbrowser import open as open_in_browser from webbrowser import open as open_in_browser
from urllib.parse import urlparse, urlunparse, parse_qsl
from logging import ( from logging import (
DEBUG, DEBUG,
INFO, INFO,
@@ -21,6 +22,7 @@ from logging import (
from orjson import loads as json_loads, JSONDecodeError from orjson import loads as json_loads, JSONDecodeError
from ._shell_signatures import Signatures_map
from scrapling import __version__ from scrapling import __version__
from scrapling.core.utils import log from scrapling.core.utils import log
from scrapling.parser import Selector, Selectors from scrapling.parser import Selector, Selectors
@@ -28,12 +30,12 @@ from scrapling.core.custom_types import TextHandler
from scrapling.engines.toolbelt.custom import Response from scrapling.engines.toolbelt.custom import Response
from scrapling.core.utils._shell import _ParseHeaders, _CookieParser from scrapling.core.utils._shell import _ParseHeaders, _CookieParser
from scrapling.core._types import ( from scrapling.core._types import (
Optional,
Dict, Dict,
Any, Any,
cast, cast,
extraction_types, Optional,
Generator, Generator,
extraction_types,
) )
@@ -312,6 +314,40 @@ class CurlParser:
return None return None
def _unpack_signature(func):
"""
Unpack TypedDict from Unpack[TypedDict] annotations in **kwargs and reconstruct the signature.
This allows the interactive shell to show individual parameters instead of just **kwargs, similar to how IDEs display them.
"""
try:
sig = signature(func)
func_name = getattr(func, "__name__", None)
# Check if this function has known parameters
if func_name not in Signatures_map:
return sig
new_params = []
for param in sig.parameters.values():
if param.kind == Parameter.VAR_KEYWORD:
# Replace **kwargs with individual keyword-only parameters
for field_name, field_type in Signatures_map[func_name].items():
new_params.append(
Parameter(field_name, Parameter.KEYWORD_ONLY, default=Parameter.empty, annotation=field_type)
)
else:
new_params.append(param)
# Reconstruct signature with unpacked parameters
if len(new_params) != len(sig.parameters):
return sig.replace(parameters=new_params)
return sig
except Exception: # pragma: no cover
return signature(func)
def show_page_in_browser(page: Selector): # pragma: no cover def show_page_in_browser(page: Selector): # pragma: no cover
if not page or not isinstance(page, Selector): if not page or not isinstance(page, Selector):
log.error("Input must be of type `Selector`") log.error("Input must be of type `Selector`")
@@ -421,7 +457,7 @@ Type 'exit' or press Ctrl+D to exit.
if isinstance(result, (Response, Selector)): if isinstance(result, (Response, Selector)):
self.pages.append(result) self.pages.append(result)
if len(self.pages) > 5: if len(self.pages) > 5:
self.pages.pop(0) # Remove oldest item self.pages.pop(0) # Remove the oldest item
# Update in IPython namespace too # Update in IPython namespace too
if self.shell: if self.shell:
@@ -431,7 +467,7 @@ Type 'exit' or press Ctrl+D to exit.
return result return result
def create_wrapper(self, func): def create_wrapper(self, func, get_signature=True):
"""Create a wrapper that preserves function signature but updates page""" """Create a wrapper that preserves function signature but updates page"""
@wraps(func) @wraps(func)
@@ -439,6 +475,12 @@ Type 'exit' or press Ctrl+D to exit.
result = func(*args, **kwargs) result = func(*args, **kwargs)
return self.update_page(result) return self.update_page(result)
if get_signature:
# Explicitly preserve and unpack signature for IPython introspection and autocompletion
wrapper.__signature__ = _unpack_signature(func) # pyright: ignore
else:
wrapper.__signature__ = signature(func) # pyright: ignore
return wrapper return wrapper
def get_namespace(self): def get_namespace(self):
@@ -451,7 +493,7 @@ Type 'exit' or press Ctrl+D to exit.
delete = self.create_wrapper(self.__Fetcher.delete) delete = self.create_wrapper(self.__Fetcher.delete)
dynamic_fetch = self.create_wrapper(self.__DynamicFetcher.fetch) dynamic_fetch = self.create_wrapper(self.__DynamicFetcher.fetch)
stealthy_fetch = self.create_wrapper(self.__StealthyFetcher.fetch) stealthy_fetch = self.create_wrapper(self.__StealthyFetcher.fetch)
curl2fetcher = self.create_wrapper(self._curl_parser.convert2fetcher) curl2fetcher = self.create_wrapper(self._curl_parser.convert2fetcher, get_signature=False)
# Create the namespace dictionary # Create the namespace dictionary
return { return {
+99 -94
View File
@@ -2,15 +2,15 @@ from time import time
from asyncio import sleep as asyncio_sleep, Lock from asyncio import sleep as asyncio_sleep, Lock
from camoufox import DefaultAddons from camoufox import DefaultAddons
from playwright.sync_api._generated import Page
from playwright.sync_api import ( from playwright.sync_api import (
Page,
Frame, Frame,
BrowserContext, BrowserContext,
Playwright, Playwright,
Response as SyncPlaywrightResponse, Response as SyncPlaywrightResponse,
) )
from playwright.async_api._generated import Page as AsyncPage
from playwright.async_api import ( from playwright.async_api import (
Page as AsyncPage,
Frame as AsyncFrame, Frame as AsyncFrame,
Playwright as AsyncPlaywright, Playwright as AsyncPlaywright,
Response as AsyncPlaywrightResponse, Response as AsyncPlaywrightResponse,
@@ -70,7 +70,7 @@ class SyncSession:
timeout: int | float, timeout: int | float,
extra_headers: Optional[Dict[str, str]], extra_headers: Optional[Dict[str, str]],
disable_resources: bool, disable_resources: bool,
) -> PageInfo: # pragma: no cover ) -> PageInfo[Page]: # pragma: no cover
"""Get a new page to use""" """Get a new page to use"""
# No need to check if a page is available or not in sync code because the code blocked before reaching here till the page closed, ofc. # No need to check if a page is available or not in sync code because the code blocked before reaching here till the page closed, ofc.
@@ -116,7 +116,7 @@ class SyncSession:
self._wait_for_networkidle(page) self._wait_for_networkidle(page)
@staticmethod @staticmethod
def _create_response_handler(page_info: PageInfo, response_container: List) -> Callable: def _create_response_handler(page_info: PageInfo[Page], response_container: List) -> Callable:
"""Create a response handler that captures the final navigation response. """Create a response handler that captures the final navigation response.
:param page_info: The PageInfo object containing the page :param page_info: The PageInfo object containing the page
@@ -175,7 +175,7 @@ class AsyncSession:
timeout: int | float, timeout: int | float,
extra_headers: Optional[Dict[str, str]], extra_headers: Optional[Dict[str, str]],
disable_resources: bool, disable_resources: bool,
) -> PageInfo: # pragma: no cover ) -> PageInfo[AsyncPage]: # pragma: no cover
"""Get a new page to use""" """Get a new page to use"""
if TYPE_CHECKING: if TYPE_CHECKING:
assert self.context is not None, "Browser context not initialized" assert self.context is not None, "Browser context not initialized"
@@ -232,7 +232,7 @@ class AsyncSession:
await self._wait_for_networkidle(page) await self._wait_for_networkidle(page)
@staticmethod @staticmethod
def _create_response_handler(page_info: PageInfo, response_container: List) -> Callable: def _create_response_handler(page_info: PageInfo[AsyncPage], response_container: List) -> Callable:
"""Create an async response handler that captures the final navigation response. """Create an async response handler that captures the final navigation response.
:param page_info: The PageInfo object containing the page :param page_info: The PageInfo object containing the page
@@ -253,130 +253,135 @@ class AsyncSession:
class DynamicSessionMixin: class DynamicSessionMixin:
def __validate__(self, **params): def __validate__(self, **params):
if "__max_pages" in params:
params["max_pages"] = params.pop("__max_pages")
config = validate(params, model=PlaywrightConfig) config = validate(params, model=PlaywrightConfig)
self.max_pages = config.max_pages self._max_pages = config.max_pages
self.headless = config.headless self._headless = config.headless
self.hide_canvas = config.hide_canvas self._hide_canvas = config.hide_canvas
self.disable_webgl = config.disable_webgl self._disable_webgl = config.disable_webgl
self.real_chrome = config.real_chrome self._real_chrome = config.real_chrome
self.stealth = config.stealth self._stealth = config.stealth
self.google_search = config.google_search self._google_search = config.google_search
self.wait = config.wait self._wait = config.wait
self.proxy = config.proxy self._proxy = config.proxy
self.locale = config.locale self._locale = config.locale
self.extra_headers = config.extra_headers self._extra_headers = config.extra_headers
self.useragent = config.useragent self._useragent = config.useragent
self.timeout = config.timeout self._timeout = config.timeout
self.cookies = config.cookies self._cookies = config.cookies
self.disable_resources = config.disable_resources self._disable_resources = config.disable_resources
self.cdp_url = config.cdp_url self._cdp_url = config.cdp_url
self.network_idle = config.network_idle self._network_idle = config.network_idle
self.load_dom = config.load_dom self._load_dom = config.load_dom
self.wait_selector = config.wait_selector self._wait_selector = config.wait_selector
self.init_script = config.init_script self._init_script = config.init_script
self.wait_selector_state = config.wait_selector_state self._wait_selector_state = config.wait_selector_state
self.extra_flags = config.extra_flags self._extra_flags = config.extra_flags
self.selector_config = config.selector_config self._selector_config = config.selector_config
self.additional_args = config.additional_args self._additional_args = config.additional_args
self.page_action = config.page_action self._page_action = config.page_action
self.user_data_dir = config.user_data_dir self._user_data_dir = config.user_data_dir
self._headers_keys = {header.lower() for header in self.extra_headers.keys()} if self.extra_headers else set() self._headers_keys = {header.lower() for header in self._extra_headers.keys()} if self._extra_headers else set()
self.__initiate_browser_options__() self.__initiate_browser_options__()
def __initiate_browser_options__(self): def __initiate_browser_options__(self):
if TYPE_CHECKING: if TYPE_CHECKING:
assert isinstance(self.proxy, tuple) assert isinstance(self._proxy, tuple)
if not self.cdp_url: if not self._cdp_url:
# `launch_options` is used with persistent context # `launch_options` is used with persistent context
self.launch_options = dict( self.launch_options = dict(
_launch_kwargs( _launch_kwargs(
self.headless, self._headless,
self.proxy, self._proxy,
self.locale, self._locale,
tuple(self.extra_headers.items()) if self.extra_headers else tuple(), tuple(self._extra_headers.items()) if self._extra_headers else tuple(),
self.useragent, self._useragent,
self.real_chrome, self._real_chrome,
self.stealth, self._stealth,
self.hide_canvas, self._hide_canvas,
self.disable_webgl, self._disable_webgl,
tuple(self.extra_flags) if self.extra_flags else tuple(), tuple(self._extra_flags) if self._extra_flags else tuple(),
) )
) )
self.launch_options["extra_http_headers"] = dict(self.launch_options["extra_http_headers"]) self.launch_options["extra_http_headers"] = dict(self.launch_options["extra_http_headers"])
self.launch_options["proxy"] = dict(self.launch_options["proxy"]) or None self.launch_options["proxy"] = dict(self.launch_options["proxy"]) or None
self.launch_options["user_data_dir"] = self.user_data_dir self.launch_options["user_data_dir"] = self._user_data_dir
self.launch_options.update(cast(Dict, self.additional_args)) self.launch_options.update(cast(Dict, self._additional_args))
self.context_options = dict() self.context_options = dict()
else: else:
# while `context_options` is left to be used when cdp mode is enabled # while `context_options` is left to be used when cdp mode is enabled
self.launch_options = dict() self.launch_options = dict()
self.context_options = dict( self.context_options = dict(
_context_kwargs( _context_kwargs(
self.proxy, self._proxy,
self.locale, self._locale,
tuple(self.extra_headers.items()) if self.extra_headers else tuple(), tuple(self._extra_headers.items()) if self._extra_headers else tuple(),
self.useragent, self._useragent,
self.stealth, self._stealth,
) )
) )
self.context_options["extra_http_headers"] = dict(self.context_options["extra_http_headers"]) self.context_options["extra_http_headers"] = dict(self.context_options["extra_http_headers"])
self.context_options["proxy"] = dict(self.context_options["proxy"]) or None self.context_options["proxy"] = dict(self.context_options["proxy"]) or None
self.context_options.update(cast(Dict, self.additional_args)) self.context_options.update(cast(Dict, self._additional_args))
class StealthySessionMixin: class StealthySessionMixin:
def __validate__(self, **params): def __validate__(self, **params):
if "__max_pages" in params:
params["max_pages"] = params.pop("__max_pages")
config: CamoufoxConfig = validate(params, model=CamoufoxConfig) config: CamoufoxConfig = validate(params, model=CamoufoxConfig)
self.max_pages = config.max_pages self._max_pages = config.max_pages
self.headless = config.headless self._headless = config.headless
self.block_images = config.block_images self._block_images = config.block_images
self.disable_resources = config.disable_resources self._disable_resources = config.disable_resources
self.block_webrtc = config.block_webrtc self._block_webrtc = config.block_webrtc
self.allow_webgl = config.allow_webgl self._allow_webgl = config.allow_webgl
self.network_idle = config.network_idle self._network_idle = config.network_idle
self.load_dom = config.load_dom self._load_dom = config.load_dom
self.humanize = config.humanize self._humanize = config.humanize
self.solve_cloudflare = config.solve_cloudflare self._solve_cloudflare = config.solve_cloudflare
self.wait = config.wait self._wait = config.wait
self.timeout = config.timeout self._timeout = config.timeout
self.page_action = config.page_action self._page_action = config.page_action
self.wait_selector = config.wait_selector self._wait_selector = config.wait_selector
self.init_script = config.init_script self._init_script = config.init_script
self.addons = config.addons self._addons = config.addons
self.wait_selector_state = config.wait_selector_state self._wait_selector_state = config.wait_selector_state
self.cookies = config.cookies self._cookies = config.cookies
self.google_search = config.google_search self._google_search = config.google_search
self.extra_headers = config.extra_headers self._extra_headers = config.extra_headers
self.proxy = config.proxy self._proxy = config.proxy
self.os_randomize = config.os_randomize self._os_randomize = config.os_randomize
self.disable_ads = config.disable_ads self._disable_ads = config.disable_ads
self.geoip = config.geoip self._geoip = config.geoip
self.selector_config = config.selector_config self._selector_config = config.selector_config
self.additional_args = config.additional_args self._additional_args = config.additional_args
self.page_action = config.page_action self._user_data_dir = config.user_data_dir
self.user_data_dir = config.user_data_dir self._headers_keys = {header.lower() for header in self._extra_headers.keys()} if self._extra_headers else set()
self._headers_keys = {header.lower() for header in self.extra_headers.keys()} if self.extra_headers else set()
self.__initiate_browser_options__() self.__initiate_browser_options__()
def __initiate_browser_options__(self): def __initiate_browser_options__(self):
"""Initiate browser options.""" """Initiate browser options."""
self.launch_options: Dict[str, Any] = generate_launch_options( self.launch_options: Dict[str, Any] = generate_launch_options(
**{ **{
"geoip": self.geoip, "geoip": self._geoip,
"proxy": dict(self.proxy) if self.proxy and isinstance(self.proxy, tuple) else self.proxy, "proxy": dict(self._proxy) if self._proxy and isinstance(self._proxy, tuple) else self._proxy,
"addons": self.addons, "addons": self._addons,
"exclude_addons": [] if self.disable_ads else [DefaultAddons.UBO], "exclude_addons": [] if self._disable_ads else [DefaultAddons.UBO],
"headless": self.headless, "headless": self._headless,
"humanize": True if self.solve_cloudflare else self.humanize, "humanize": True if self._solve_cloudflare else self._humanize,
"i_know_what_im_doing": True, # To turn warnings off with the user configurations "i_know_what_im_doing": True, # To turn warnings off with the user configurations
"allow_webgl": self.allow_webgl, "allow_webgl": self._allow_webgl,
"block_webrtc": self.block_webrtc, "block_webrtc": self._block_webrtc,
"block_images": self.block_images, # Careful! it makes some websites don't finish loading at all like stackoverflow even in headful mode. "block_images": self._block_images, # Careful! it makes some websites don't finish loading at all like stackoverflow even in headful mode.
"os": None if self.os_randomize else get_os_name(), "os": None if self._os_randomize else get_os_name(),
"user_data_dir": self.user_data_dir, "user_data_dir": self._user_data_dir,
"ff_version": __ff_version_str__, "ff_version": __ff_version_str__,
"firefox_user_prefs": { "firefox_user_prefs": {
# This is what enabling `enable_cache` does internally, so we do it from here instead # This is what enabling `enable_cache` does internally, so we do it from here instead
@@ -386,7 +391,7 @@ class StealthySessionMixin:
"browser.cache.disk_cache_ssl": True, "browser.cache.disk_cache_ssl": True,
"browser.cache.disk.smart_size.enabled": True, "browser.cache.disk.smart_size.enabled": True,
}, },
**cast(Dict, self.additional_args), **cast(Dict, self._additional_args),
} }
) )
+84 -271
View File
@@ -14,58 +14,47 @@ from playwright.async_api import (
BrowserContext as AsyncBrowserContext, BrowserContext as AsyncBrowserContext,
) )
from ._validators import validate_fetch as _validate, CamoufoxConfig
from ._base import SyncSession, AsyncSession, StealthySessionMixin
from scrapling.core.utils import log from scrapling.core.utils import log
from scrapling.core._types import ( from ._types import CamoufoxSession, CamoufoxFetchParams
Any, from scrapling.core._types import Any, Unpack, TYPE_CHECKING
Dict, from ._base import SyncSession, AsyncSession, StealthySessionMixin
List, from ._validators import validate_fetch as _validate, CamoufoxConfig
Optional, from scrapling.engines.toolbelt.convertor import Response, ResponseFactory
Callable,
TYPE_CHECKING,
SelectorWaitStates,
)
from scrapling.engines.toolbelt.convertor import (
Response,
ResponseFactory,
)
from scrapling.engines.toolbelt.fingerprints import generate_convincing_referer from scrapling.engines.toolbelt.fingerprints import generate_convincing_referer
__CF_PATTERN__ = re_compile("challenges.cloudflare.com/cdn-cgi/challenge-platform/.*") __CF_PATTERN__ = re_compile("challenges.cloudflare.com/cdn-cgi/challenge-platform/.*")
_UNSET: Any = object()
class StealthySession(StealthySessionMixin, SyncSession): class StealthySession(StealthySessionMixin, SyncSession):
"""A Stealthy session manager with page pooling.""" """A Stealthy session manager with page pooling."""
__slots__ = ( __slots__ = (
"max_pages", "_max_pages",
"headless", "_headless",
"block_images", "_block_images",
"disable_resources", "_disable_resources",
"block_webrtc", "_block_webrtc",
"allow_webgl", "_allow_webgl",
"network_idle", "_network_idle",
"load_dom", "_load_dom",
"humanize", "_humanize",
"solve_cloudflare", "_solve_cloudflare",
"wait", "_wait",
"timeout", "_timeout",
"page_action", "_page_action",
"wait_selector", "_wait_selector",
"init_script", "_init_script",
"addons", "_addons",
"wait_selector_state", "_wait_selector_state",
"cookies", "_cookies",
"google_search", "_google_search",
"extra_headers", "_extra_headers",
"proxy", "_proxy",
"os_randomize", "_os_randomize",
"disable_ads", "_disable_ads",
"geoip", "_geoip",
"selector_config", "_selector_config",
"additional_args", "_additional_args",
"playwright", "playwright",
"browser", "browser",
"context", "context",
@@ -73,38 +62,10 @@ class StealthySession(StealthySessionMixin, SyncSession):
"_closed", "_closed",
"launch_options", "launch_options",
"_headers_keys", "_headers_keys",
"_user_data_dir",
) )
def __init__( def __init__(self, **kwargs: Unpack[CamoufoxSession]):
self,
__max_pages: int = 1,
headless: bool = True, # noqa: F821
block_images: bool = False,
disable_resources: bool = False,
block_webrtc: bool = False,
allow_webgl: bool = True,
network_idle: bool = False,
load_dom: bool = True,
humanize: bool | float = True,
solve_cloudflare: bool = False,
wait: int | float = 0,
timeout: int | float = 30000,
page_action: Optional[Callable] = None,
wait_selector: Optional[str] = None,
init_script: Optional[str] = None,
addons: Optional[List[str]] = None,
wait_selector_state: SelectorWaitStates = "attached",
cookies: Optional[List[Dict]] = None,
google_search: bool = True,
extra_headers: Optional[Dict[str, str]] = None,
proxy: Optional[str | Dict[str, str]] = None,
os_randomize: bool = False,
disable_ads: bool = False,
geoip: bool = False,
user_data_dir: str = "",
selector_config: Optional[Dict] = None,
additional_args: Optional[Dict] = None,
):
"""A Browser session manager with page pooling """A Browser session manager with page pooling
:param headless: Run the browser in headless/hidden (default), or headful/visible mode. :param headless: Run the browser in headless/hidden (default), or headful/visible mode.
@@ -138,50 +99,21 @@ class StealthySession(StealthySessionMixin, SyncSession):
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. :param selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param additional_args: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings. :param additional_args: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings.
""" """
self.__validate__(**kwargs)
self.__validate__( super().__init__(max_pages=self._max_pages)
wait=wait,
proxy=proxy,
geoip=geoip,
addons=addons,
timeout=timeout,
cookies=cookies,
headless=headless,
humanize=humanize,
load_dom=load_dom,
max_pages=__max_pages,
disable_ads=disable_ads,
allow_webgl=allow_webgl,
page_action=page_action,
init_script=init_script,
network_idle=network_idle,
block_images=block_images,
block_webrtc=block_webrtc,
os_randomize=os_randomize,
user_data_dir=user_data_dir,
wait_selector=wait_selector,
google_search=google_search,
extra_headers=extra_headers,
additional_args=additional_args,
selector_config=selector_config,
solve_cloudflare=solve_cloudflare,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
)
super().__init__(max_pages=self.max_pages)
def __create__(self): def __create__(self):
"""Create a browser for this instance and context.""" """Create a browser for this instance and context."""
self.playwright = sync_playwright().start() self.playwright = sync_playwright().start()
self.context = self.playwright.firefox.launch_persistent_context(**self.launch_options) self.context = self.playwright.firefox.launch_persistent_context(**self.launch_options)
if self.init_script: # pragma: no cover if self._init_script: # pragma: no cover
self.context.add_init_script(path=self.init_script) self.context.add_init_script(path=self._init_script)
if self.cookies: # pragma: no cover if self._cookies: # pragma: no cover
self.context.add_cookies(self.cookies) self.context.add_cookies(self._cookies)
def _solve_cloudflare(self, page: Page) -> None: # pragma: no cover def _cloudflare_solver(self, page: Page) -> None: # pragma: no cover
"""Solve the cloudflare challenge displayed on the playwright page passed """Solve the cloudflare challenge displayed on the playwright page passed
:param page: The targeted page :param page: The targeted page
@@ -247,59 +179,28 @@ class StealthySession(StealthySessionMixin, SyncSession):
log.info("Cloudflare captcha is solved") log.info("Cloudflare captcha is solved")
return return
def fetch( def fetch(self, url: str, **kwargs: Unpack[CamoufoxFetchParams]) -> Response:
self,
url: str,
google_search: bool = _UNSET,
timeout: int | float = _UNSET,
wait: int | float = _UNSET,
page_action: Optional[Callable] = _UNSET,
extra_headers: Optional[Dict[str, str]] = _UNSET,
disable_resources: bool = _UNSET,
wait_selector: Optional[str] = _UNSET,
wait_selector_state: SelectorWaitStates = _UNSET,
network_idle: bool = _UNSET,
load_dom: bool = _UNSET,
solve_cloudflare: bool = _UNSET,
selector_config: Optional[Dict] = _UNSET,
) -> Response:
"""Opens up the browser and do your request based on your chosen options. """Opens up the browser and do your request based on your chosen options.
:param url: The Target url. :param url: The Target url.
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. :param kwargs: Additional keyword arguments including:
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - page_action: Added for automation. A function that takes the `page` object and does the automation you need.
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. - disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
:param wait_selector: Wait for a specific CSS selector to be in a specific state. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - wait_selector: Wait for a specific CSS selector to be in a specific state.
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
:param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you. - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute.
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. - solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you.
- selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:return: A `Response` object. :return: A `Response` object.
""" """
params = _validate( params = _validate(kwargs, self, CamoufoxConfig)
[
("google_search", google_search, self.google_search),
("timeout", timeout, self.timeout),
("wait", wait, self.wait),
("page_action", page_action, self.page_action),
("extra_headers", extra_headers, self.extra_headers),
("disable_resources", disable_resources, self.disable_resources),
("wait_selector", wait_selector, self.wait_selector),
("wait_selector_state", wait_selector_state, self.wait_selector_state),
("network_idle", network_idle, self.network_idle),
("load_dom", load_dom, self.load_dom),
("solve_cloudflare", solve_cloudflare, self.solve_cloudflare),
("selector_config", selector_config, self.selector_config),
],
CamoufoxConfig,
_UNSET,
)
if self._closed: # pragma: no cover if self._closed: # pragma: no cover
raise RuntimeError("Context manager has been closed") raise RuntimeError("Context manager has been closed")
@@ -322,7 +223,7 @@ class StealthySession(StealthySessionMixin, SyncSession):
raise RuntimeError(f"Failed to get response for {url}") raise RuntimeError(f"Failed to get response for {url}")
if params.solve_cloudflare: if params.solve_cloudflare:
self._solve_cloudflare(page_info.page) self._cloudflare_solver(page_info.page)
# Make sure the page is fully loaded after the captcha # Make sure the page is fully loaded after the captcha
self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle) self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
@@ -360,36 +261,7 @@ class StealthySession(StealthySessionMixin, SyncSession):
class AsyncStealthySession(StealthySessionMixin, AsyncSession): class AsyncStealthySession(StealthySessionMixin, AsyncSession):
"""A Stealthy session manager with page pooling.""" """A Stealthy session manager with page pooling."""
def __init__( def __init__(self, **kwargs: Unpack[CamoufoxSession]):
self,
max_pages: int = 1,
headless: bool = True, # noqa: F821
block_images: bool = False,
disable_resources: bool = False,
block_webrtc: bool = False,
allow_webgl: bool = True,
network_idle: bool = False,
load_dom: bool = True,
humanize: bool | float = True,
solve_cloudflare: bool = False,
wait: int | float = 0,
timeout: int | float = 30000,
page_action: Optional[Callable] = None,
wait_selector: Optional[str] = None,
init_script: Optional[str] = None,
addons: Optional[List[str]] = None,
wait_selector_state: SelectorWaitStates = "attached",
cookies: Optional[List[Dict]] = None,
google_search: bool = True,
extra_headers: Optional[Dict[str, str]] = None,
proxy: Optional[str | Dict[str, str]] = None,
os_randomize: bool = False,
disable_ads: bool = False,
geoip: bool = False,
user_data_dir: str = "",
selector_config: Optional[Dict] = None,
additional_args: Optional[Dict] = None,
):
"""A Browser session manager with page pooling """A Browser session manager with page pooling
:param headless: Run the browser in headless/hidden (default), or headful/visible mode. :param headless: Run the browser in headless/hidden (default), or headful/visible mode.
@@ -424,36 +296,8 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. :param selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param additional_args: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings. :param additional_args: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings.
""" """
self.__validate__( self.__validate__(**kwargs)
wait=wait, super().__init__(max_pages=self._max_pages)
proxy=proxy,
geoip=geoip,
addons=addons,
timeout=timeout,
cookies=cookies,
headless=headless,
load_dom=load_dom,
humanize=humanize,
max_pages=max_pages,
disable_ads=disable_ads,
allow_webgl=allow_webgl,
page_action=page_action,
init_script=init_script,
network_idle=network_idle,
block_images=block_images,
block_webrtc=block_webrtc,
os_randomize=os_randomize,
wait_selector=wait_selector,
google_search=google_search,
extra_headers=extra_headers,
user_data_dir=user_data_dir,
additional_args=additional_args,
selector_config=selector_config,
solve_cloudflare=solve_cloudflare,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
)
super().__init__(max_pages=self.max_pages)
async def __create__(self): async def __create__(self):
"""Create a browser for this instance and context.""" """Create a browser for this instance and context."""
@@ -462,13 +306,13 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
**self.launch_options **self.launch_options
) )
if self.init_script: # pragma: no cover if self._init_script: # pragma: no cover
await self.context.add_init_script(path=self.init_script) await self.context.add_init_script(path=self._init_script)
if self.cookies: if self._cookies:
await self.context.add_cookies(self.cookies) # pyright: ignore [reportArgumentType] await self.context.add_cookies(self._cookies) # pyright: ignore [reportArgumentType]
async def _solve_cloudflare(self, page: async_Page): # pragma: no cover async def _cloudflare_solver(self, page: async_Page): # pragma: no cover
"""Solve the cloudflare challenge displayed on the playwright page passed. The async version """Solve the cloudflare challenge displayed on the playwright page passed. The async version
:param page: The async targeted page :param page: The async targeted page
@@ -534,59 +378,28 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
log.info("Cloudflare captcha is solved") log.info("Cloudflare captcha is solved")
return return
async def fetch( async def fetch(self, url: str, **kwargs: Unpack[CamoufoxFetchParams]) -> Response:
self,
url: str,
google_search: bool = _UNSET,
timeout: int | float = _UNSET,
wait: int | float = _UNSET,
page_action: Optional[Callable] = _UNSET,
extra_headers: Optional[Dict[str, str]] = _UNSET,
disable_resources: bool = _UNSET,
wait_selector: Optional[str] = _UNSET,
wait_selector_state: SelectorWaitStates = _UNSET,
network_idle: bool = _UNSET,
load_dom: bool = _UNSET,
solve_cloudflare: bool = _UNSET,
selector_config: Optional[Dict] = _UNSET,
) -> Response:
"""Opens up the browser and do your request based on your chosen options. """Opens up the browser and do your request based on your chosen options.
:param url: The Target url. :param url: The Target url.
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. :param kwargs: Additional keyword arguments including:
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - page_action: Added for automation. A function that takes the `page` object and does the automation you need.
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. - disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
:param wait_selector: Wait for a specific CSS selector to be in a specific state. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - wait_selector: Wait for a specific CSS selector to be in a specific state.
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
:param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you. - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute.
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. - solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you.
- selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:return: A `Response` object. :return: A `Response` object.
""" """
params = _validate( params = _validate(kwargs, self, CamoufoxConfig)
[
("google_search", google_search, self.google_search),
("timeout", timeout, self.timeout),
("wait", wait, self.wait),
("page_action", page_action, self.page_action),
("extra_headers", extra_headers, self.extra_headers),
("disable_resources", disable_resources, self.disable_resources),
("wait_selector", wait_selector, self.wait_selector),
("wait_selector_state", wait_selector_state, self.wait_selector_state),
("network_idle", network_idle, self.network_idle),
("load_dom", load_dom, self.load_dom),
("solve_cloudflare", solve_cloudflare, self.solve_cloudflare),
("selector_config", selector_config, self.selector_config),
],
CamoufoxConfig,
_UNSET,
)
if self._closed: # pragma: no cover if self._closed: # pragma: no cover
raise RuntimeError("Context manager has been closed") raise RuntimeError("Context manager has been closed")
@@ -613,7 +426,7 @@ class AsyncStealthySession(StealthySessionMixin, AsyncSession):
raise RuntimeError(f"Failed to get response for {url}") raise RuntimeError(f"Failed to get response for {url}")
if params.solve_cloudflare: if params.solve_cloudflare:
await self._solve_cloudflare(page_info.page) await self._cloudflare_solver(page_info.page)
# Make sure the page is fully loaded after the captcha # Make sure the page is fully loaded after the captcha
await self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle) await self._wait_for_page_stability(page_info.page, params.load_dom, params.network_idle)
+81 -259
View File
@@ -13,92 +13,55 @@ from patchright.sync_api import sync_playwright as sync_patchright
from patchright.async_api import async_playwright as async_patchright from patchright.async_api import async_playwright as async_patchright
from scrapling.core.utils import log from scrapling.core.utils import log
from scrapling.core._types import Unpack, TYPE_CHECKING
from ._types import PlaywrightSession, PlaywrightFetchParams
from ._base import SyncSession, AsyncSession, DynamicSessionMixin from ._base import SyncSession, AsyncSession, DynamicSessionMixin
from ._validators import validate_fetch as _validate, PlaywrightConfig from ._validators import validate_fetch as _validate, PlaywrightConfig
from scrapling.core._types import ( from scrapling.engines.toolbelt.convertor import Response, ResponseFactory
Any,
Dict,
List,
Optional,
Callable,
TYPE_CHECKING,
SelectorWaitStates,
)
from scrapling.engines.toolbelt.convertor import (
Response,
ResponseFactory,
)
from scrapling.engines.toolbelt.fingerprints import generate_convincing_referer from scrapling.engines.toolbelt.fingerprints import generate_convincing_referer
_UNSET: Any = object()
class DynamicSession(DynamicSessionMixin, SyncSession): class DynamicSession(DynamicSessionMixin, SyncSession):
"""A Browser session manager with page pooling.""" """A Browser session manager with page pooling."""
__slots__ = ( __slots__ = (
"max_pages", "_max_pages",
"headless", "_headless",
"hide_canvas", "_hide_canvas",
"disable_webgl", "_disable_webgl",
"real_chrome", "_real_chrome",
"stealth", "_stealth",
"google_search", "_google_search",
"proxy", "_proxy",
"locale", "_locale",
"extra_headers", "_extra_headers",
"useragent", "_useragent",
"timeout", "_timeout",
"cookies", "_cookies",
"disable_resources", "_disable_resources",
"network_idle", "_network_idle",
"load_dom", "_load_dom",
"wait_selector", "_wait_selector",
"init_script", "_init_script",
"wait_selector_state", "_wait_selector_state",
"wait", "_wait",
"playwright", "playwright",
"browser", "browser",
"context", "context",
"page_pool", "page_pool",
"_closed", "_closed",
"selector_config", "_selector_config",
"page_action", "_page_action",
"launch_options", "launch_options",
"context_options", "context_options",
"cdp_url", "_cdp_url",
"_headers_keys", "_headers_keys",
"_extra_flags",
"_additional_args",
"_user_data_dir",
) )
def __init__( def __init__(self, **kwargs: Unpack[PlaywrightSession]):
self,
__max_pages: int = 1,
headless: bool = True,
google_search: bool = True,
hide_canvas: bool = False,
disable_webgl: bool = False,
real_chrome: bool = False,
stealth: bool = False,
wait: int | float = 0,
page_action: Optional[Callable] = None,
proxy: Optional[str | Dict[str, str]] = None,
locale: str = "en-US",
extra_headers: Optional[Dict[str, str]] = None,
useragent: Optional[str] = None,
cdp_url: Optional[str] = None,
timeout: int | float = 30000,
disable_resources: bool = False,
wait_selector: Optional[str] = None,
init_script: Optional[str] = None,
cookies: Optional[List[Dict]] = None,
network_idle: bool = False,
load_dom: bool = True,
wait_selector_state: SelectorWaitStates = "attached",
user_data_dir: str = "",
extra_flags: Optional[List[str]] = None,
selector_config: Optional[Dict] = None,
additional_args: Optional[Dict] = None,
):
"""A Browser session manager with page pooling, it's using a persistent browser Context by default with a temporary user profile directory. """A Browser session manager with page pooling, it's using a persistent browser Context by default with a temporary user profile directory.
:param headless: Run the browser in headless/hidden (default), or headful/visible mode. :param headless: Run the browser in headless/hidden (default), or headful/visible mode.
@@ -129,105 +92,49 @@ class DynamicSession(DynamicSessionMixin, SyncSession):
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. :param selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings. :param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings.
""" """
self.__validate__( self.__validate__(**kwargs)
wait=wait, super().__init__(max_pages=self._max_pages)
proxy=proxy,
locale=locale,
timeout=timeout,
stealth=stealth,
cdp_url=cdp_url,
cookies=cookies,
load_dom=load_dom,
headless=headless,
useragent=useragent,
max_pages=__max_pages,
real_chrome=real_chrome,
page_action=page_action,
hide_canvas=hide_canvas,
init_script=init_script,
network_idle=network_idle,
user_data_dir=user_data_dir,
google_search=google_search,
extra_headers=extra_headers,
wait_selector=wait_selector,
disable_webgl=disable_webgl,
extra_flags=extra_flags,
selector_config=selector_config,
additional_args=additional_args,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
)
super().__init__(max_pages=self.max_pages)
def __create__(self): def __create__(self):
"""Create a browser for this instance and context.""" """Create a browser for this instance and context."""
sync_context = sync_patchright if self.stealth else sync_playwright sync_context = sync_patchright if self._stealth else sync_playwright
self.playwright: Playwright = sync_context().start() # pyright: ignore [reportAttributeAccessIssue] self.playwright: Playwright = sync_context().start() # pyright: ignore [reportAttributeAccessIssue]
if self.cdp_url: # pragma: no cover if self._cdp_url: # pragma: no cover
self.context = self.playwright.chromium.connect_over_cdp(endpoint_url=self.cdp_url).new_context( self.context = self.playwright.chromium.connect_over_cdp(endpoint_url=self._cdp_url).new_context(
**self.context_options **self.context_options
) )
else: else:
self.context = self.playwright.chromium.launch_persistent_context(**self.launch_options) self.context = self.playwright.chromium.launch_persistent_context(**self.launch_options)
if self.init_script: # pragma: no cover if self._init_script: # pragma: no cover
self.context.add_init_script(path=self.init_script) self.context.add_init_script(path=self._init_script)
if self.cookies: # pragma: no cover if self._cookies: # pragma: no cover
self.context.add_cookies(self.cookies) self.context.add_cookies(self._cookies)
def fetch( def fetch(self, url: str, **kwargs: Unpack[PlaywrightFetchParams]) -> Response:
self,
url: str,
google_search: bool = _UNSET,
timeout: int | float = _UNSET,
wait: int | float = _UNSET,
page_action: Optional[Callable] = _UNSET,
extra_headers: Optional[Dict[str, str]] = _UNSET,
disable_resources: bool = _UNSET,
wait_selector: Optional[str] = _UNSET,
wait_selector_state: SelectorWaitStates = _UNSET,
network_idle: bool = _UNSET,
load_dom: bool = _UNSET,
selector_config: Optional[Dict] = _UNSET,
) -> Response:
"""Opens up the browser and do your request based on your chosen options. """Opens up the browser and do your request based on your chosen options.
:param url: The Target url. :param url: The Target url.
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. :param kwargs: Additional keyword arguments including:
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - page_action: Added for automation. A function that takes the `page` object and does the automation you need.
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. - disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
:param wait_selector: Wait for a specific CSS selector to be in a specific state. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - wait_selector: Wait for a specific CSS selector to be in a specific state.
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
:param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute.
- selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:return: A `Response` object. :return: A `Response` object.
""" """
params = _validate( params = _validate(kwargs, self, PlaywrightConfig)
[
("google_search", google_search, self.google_search),
("timeout", timeout, self.timeout),
("wait", wait, self.wait),
("page_action", page_action, self.page_action),
("extra_headers", extra_headers, self.extra_headers),
("disable_resources", disable_resources, self.disable_resources),
("wait_selector", wait_selector, self.wait_selector),
("wait_selector_state", wait_selector_state, self.wait_selector_state),
("network_idle", network_idle, self.network_idle),
("load_dom", load_dom, self.load_dom),
("selector_config", selector_config, self.selector_config),
],
PlaywrightConfig,
_UNSET,
)
if self._closed: # pragma: no cover if self._closed: # pragma: no cover
raise RuntimeError("Context manager has been closed") raise RuntimeError("Context manager has been closed")
@@ -285,35 +192,7 @@ class DynamicSession(DynamicSessionMixin, SyncSession):
class AsyncDynamicSession(DynamicSessionMixin, AsyncSession): class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
"""An async Browser session manager with page pooling, it's using a persistent browser Context by default with a temporary user profile directory.""" """An async Browser session manager with page pooling, it's using a persistent browser Context by default with a temporary user profile directory."""
def __init__( def __init__(self, **kwargs: Unpack[PlaywrightSession]):
self,
max_pages: int = 1,
headless: bool = True,
google_search: bool = True,
hide_canvas: bool = False,
disable_webgl: bool = False,
real_chrome: bool = False,
stealth: bool = False,
wait: int | float = 0,
page_action: Optional[Callable] = None,
proxy: Optional[str | Dict[str, str]] = None,
locale: str = "en-US",
extra_headers: Optional[Dict[str, str]] = None,
useragent: Optional[str] = None,
cdp_url: Optional[str] = None,
timeout: int | float = 30000,
disable_resources: bool = False,
wait_selector: Optional[str] = None,
init_script: Optional[str] = None,
cookies: Optional[List[Dict]] = None,
network_idle: bool = False,
load_dom: bool = True,
wait_selector_state: SelectorWaitStates = "attached",
user_data_dir: str = "",
extra_flags: Optional[List[str]] = None,
selector_config: Optional[Dict] = None,
additional_args: Optional[Dict] = None,
):
"""A Browser session manager with page pooling """A Browser session manager with page pooling
:param headless: Run the browser in headless/hidden (default), or headful/visible mode. :param headless: Run the browser in headless/hidden (default), or headful/visible mode.
@@ -345,107 +224,50 @@ class AsyncDynamicSession(DynamicSessionMixin, AsyncSession):
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. :param selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings. :param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings.
""" """
self.__validate__(**kwargs)
self.__validate__( super().__init__(max_pages=self._max_pages)
wait=wait,
proxy=proxy,
locale=locale,
timeout=timeout,
stealth=stealth,
cdp_url=cdp_url,
cookies=cookies,
load_dom=load_dom,
headless=headless,
useragent=useragent,
max_pages=max_pages,
real_chrome=real_chrome,
page_action=page_action,
hide_canvas=hide_canvas,
init_script=init_script,
network_idle=network_idle,
user_data_dir=user_data_dir,
google_search=google_search,
extra_headers=extra_headers,
wait_selector=wait_selector,
disable_webgl=disable_webgl,
extra_flags=extra_flags,
selector_config=selector_config,
additional_args=additional_args,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
)
super().__init__(max_pages=self.max_pages)
async def __create__(self): async def __create__(self):
"""Create a browser for this instance and context.""" """Create a browser for this instance and context."""
async_context = async_patchright if self.stealth else async_playwright async_context = async_patchright if self._stealth else async_playwright
self.playwright: AsyncPlaywright = await async_context().start() # pyright: ignore [reportAttributeAccessIssue] self.playwright: AsyncPlaywright = await async_context().start() # pyright: ignore [reportAttributeAccessIssue]
if self.cdp_url: if self._cdp_url:
browser = await self.playwright.chromium.connect_over_cdp(endpoint_url=self.cdp_url) browser = await self.playwright.chromium.connect_over_cdp(endpoint_url=self._cdp_url)
self.context: AsyncBrowserContext = await browser.new_context(**self.context_options) self.context: AsyncBrowserContext = await browser.new_context(**self.context_options)
else: else:
self.context: AsyncBrowserContext = await self.playwright.chromium.launch_persistent_context( self.context: AsyncBrowserContext = await self.playwright.chromium.launch_persistent_context(
**self.launch_options **self.launch_options
) )
if self.init_script: # pragma: no cover if self._init_script: # pragma: no cover
await self.context.add_init_script(path=self.init_script) await self.context.add_init_script(path=self._init_script)
if self.cookies: if self._cookies:
await self.context.add_cookies(self.cookies) # pyright: ignore await self.context.add_cookies(self._cookies) # pyright: ignore
async def fetch( async def fetch(self, url: str, **kwargs: Unpack[PlaywrightFetchParams]) -> Response:
self,
url: str,
google_search: bool = _UNSET,
timeout: int | float = _UNSET,
wait: int | float = _UNSET,
page_action: Optional[Callable] = _UNSET,
extra_headers: Optional[Dict[str, str]] = _UNSET,
disable_resources: bool = _UNSET,
wait_selector: Optional[str] = _UNSET,
wait_selector_state: SelectorWaitStates = _UNSET,
network_idle: bool = _UNSET,
load_dom: bool = _UNSET,
selector_config: Optional[Dict] = _UNSET,
) -> Response:
"""Opens up the browser and do your request based on your chosen options. """Opens up the browser and do your request based on your chosen options.
:param url: The Target url. :param url: The Target url.
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. :param kwargs: Additional keyword arguments including:
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - page_action: Added for automation. A function that takes the `page` object and does the automation you need.
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. - disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`.
:param wait_selector: Wait for a specific CSS selector to be in a specific state. This can help save your proxy usage but be careful with this option as it makes some websites never finish loading.
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - wait_selector: Wait for a specific CSS selector to be in a specific state.
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
:param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param selector_config: The arguments that will be passed in the end while creating the final Selector's class. - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute.
- selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:return: A `Response` object. :return: A `Response` object.
""" """
params = _validate( params = _validate(kwargs, self, PlaywrightConfig)
[
("google_search", google_search, self.google_search),
("timeout", timeout, self.timeout),
("wait", wait, self.wait),
("page_action", page_action, self.page_action),
("extra_headers", extra_headers, self.extra_headers),
("disable_resources", disable_resources, self.disable_resources),
("wait_selector", wait_selector, self.wait_selector),
("wait_selector_state", wait_selector_state, self.wait_selector_state),
("network_idle", network_idle, self.network_idle),
("load_dom", load_dom, self.load_dom),
("selector_config", selector_config, self.selector_config),
],
PlaywrightConfig,
_UNSET,
)
if self._closed: # pragma: no cover if self._closed: # pragma: no cover
raise RuntimeError("Context manager has been closed") raise RuntimeError("Context manager has been closed")
+19 -8
View File
@@ -1,20 +1,21 @@
from threading import RLock from threading import RLock
from dataclasses import dataclass from dataclasses import dataclass
from playwright.sync_api import Page as SyncPage from playwright.sync_api._generated import Page as SyncPage
from playwright.async_api import Page as AsyncPage from playwright.async_api._generated import Page as AsyncPage
from scrapling.core._types import Optional, List, Literal from scrapling.core._types import Optional, List, Literal, overload, TypeVar, Generic, cast
PageState = Literal["ready", "busy", "error"] # States that a page can be in PageState = Literal["ready", "busy", "error"] # States that a page can be in
PageType = TypeVar("PageType", SyncPage, AsyncPage)
@dataclass @dataclass
class PageInfo: class PageInfo(Generic[PageType]):
"""Information about the page and its current state""" """Information about the page and its current state"""
__slots__ = ("page", "state", "url") __slots__ = ("page", "state", "url")
page: SyncPage | AsyncPage page: PageType
state: PageState state: PageState
url: Optional[str] url: Optional[str]
@@ -44,16 +45,26 @@ class PagePool:
def __init__(self, max_pages: int = 5): def __init__(self, max_pages: int = 5):
self.max_pages = max_pages self.max_pages = max_pages
self.pages: List[PageInfo] = [] self.pages: List[PageInfo[SyncPage] | PageInfo[AsyncPage]] = []
self._lock = RLock() self._lock = RLock()
def add_page(self, page: SyncPage | AsyncPage) -> PageInfo: @overload
def add_page(self, page: SyncPage) -> PageInfo[SyncPage]: ...
@overload
def add_page(self, page: AsyncPage) -> PageInfo[AsyncPage]: ...
def add_page(self, page: SyncPage | AsyncPage) -> PageInfo[SyncPage] | PageInfo[AsyncPage]:
"""Add a new page to the pool""" """Add a new page to the pool"""
with self._lock: with self._lock:
if len(self.pages) >= self.max_pages: if len(self.pages) >= self.max_pages:
raise RuntimeError(f"Maximum page limit ({self.max_pages}) reached") raise RuntimeError(f"Maximum page limit ({self.max_pages}) reached")
page_info = PageInfo(page, "ready", "") if isinstance(page, AsyncPage):
page_info = cast(PageInfo[AsyncPage], PageInfo(page, "ready", ""))
else:
page_info = cast(PageInfo[SyncPage], PageInfo(page, "ready", ""))
self.pages.append(page_info) self.pages.append(page_info)
return page_info return page_info
+120
View File
@@ -0,0 +1,120 @@
from curl_cffi.requests import (
ProxySpec,
CookieTypes,
BrowserTypeLiteral,
)
from scrapling.core._types import (
Dict,
List,
Tuple,
Mapping,
Optional,
Callable,
Iterable,
TypedDict,
TypeAlias,
SelectorWaitStates,
TYPE_CHECKING,
)
# Type alias for `impersonate` parameter - accepts a single browser or list of browsers
ImpersonateType: TypeAlias = BrowserTypeLiteral | List[BrowserTypeLiteral] | None
if TYPE_CHECKING: # pragma: no cover
# Types for session initialization
class RequestsSession(TypedDict, total=False):
impersonate: ImpersonateType
http3: Optional[bool]
stealthy_headers: Optional[bool]
proxies: Optional[ProxySpec]
proxy: Optional[str]
proxy_auth: Optional[Tuple[str, str]]
timeout: Optional[int | float]
headers: Optional[Mapping[str, Optional[str]]]
retries: Optional[int]
retry_delay: Optional[int]
follow_redirects: Optional[bool]
max_redirects: Optional[int]
verify: Optional[bool]
cert: Optional[str | Tuple[str, str]]
selector_config: Optional[Dict]
# Types for GET request method parameters
class GetRequestParams(RequestsSession, total=False):
params: Optional[Dict | List | Tuple]
cookies: Optional[CookieTypes]
auth: Optional[Tuple[str, str]]
# Types for POST/PUT/DELETE request method parameters
class DataRequestParams(GetRequestParams, total=False):
data: Optional[Dict | str]
json: Optional[Dict | List]
# Types for browser session
class BrowserSession(TypedDict, total=False):
max_pages: int
headless: bool
disable_resources: bool
network_idle: bool
load_dom: bool
wait_selector: Optional[str]
wait_selector_state: SelectorWaitStates
cookies: Optional[Iterable[Dict]]
google_search: bool
wait: int | float
page_action: Optional[Callable]
proxy: Optional[str | Dict[str, str] | Tuple]
extra_headers: Optional[Dict[str, str]]
timeout: int | float
init_script: Optional[str]
user_data_dir: str
selector_config: Optional[Dict]
additional_args: Optional[Dict]
class PlaywrightSession(BrowserSession, total=False):
cdp_url: Optional[str]
hide_canvas: bool
disable_webgl: bool
real_chrome: bool
stealth: bool
locale: str
useragent: Optional[str]
extra_flags: Optional[List[str]]
class PlaywrightFetchParams(TypedDict, total=False):
google_search: bool
timeout: int | float
wait: int | float
page_action: Optional[Callable]
extra_headers: Optional[Dict[str, str]]
disable_resources: bool
wait_selector: Optional[str]
wait_selector_state: SelectorWaitStates
network_idle: bool
load_dom: bool
selector_config: Optional[Dict]
class CamoufoxSession(BrowserSession, total=False):
block_images: bool
block_webrtc: bool
allow_webgl: bool
humanize: bool | float
solve_cloudflare: bool
addons: Optional[List[str]]
os_randomize: bool
disable_ads: bool
geoip: bool
class CamoufoxFetchParams(PlaywrightFetchParams, total=False):
solve_cloudflare: bool
else: # pragma: no cover
RequestsSession = TypedDict
GetRequestParams = TypedDict
DataRequestParams = TypedDict
PlaywrightSession = TypedDict
PlaywrightFetchParams = TypedDict
CamoufoxSession = TypedDict
CamoufoxFetchParams = TypedDict
+23 -13
View File
@@ -7,6 +7,7 @@ from dataclasses import dataclass, fields
from msgspec import Struct, Meta, convert, ValidationError from msgspec import Struct, Meta, convert, ValidationError
from scrapling.core._types import ( from scrapling.core._types import (
Any,
Dict, Dict,
List, List,
Tuple, Tuple,
@@ -17,11 +18,12 @@ from scrapling.core._types import (
overload, overload,
) )
from scrapling.engines.toolbelt.navigation import construct_proxy_dict from scrapling.engines.toolbelt.navigation import construct_proxy_dict
from scrapling.engines._browsers._types import PlaywrightFetchParams, CamoufoxFetchParams
# Custom validators for msgspec # Custom validators for msgspec
@lru_cache(8) @lru_cache(8)
def _is_invalid_file_path(value: str) -> bool | str: def _is_invalid_file_path(value: str) -> bool | str: # pragma: no cover
"""Fast file path validation""" """Fast file path validation"""
path = Path(value) path = Path(value)
if not path.exists(): if not path.exists():
@@ -33,7 +35,7 @@ def _is_invalid_file_path(value: str) -> bool | str:
return False return False
def _validate_addon_path(value: str) -> None: def _validate_addon_path(value: str) -> None: # pragma: no cover
"""Fast addon path validation""" """Fast addon path validation"""
path = Path(value) path = Path(value)
if not path.exists(): if not path.exists():
@@ -49,7 +51,7 @@ def _is_invalid_cdp_url(cdp_url: str) -> bool | str:
return "CDP URL must use 'ws://' or 'wss://' scheme" return "CDP URL must use 'ws://' or 'wss://' scheme"
netloc = urlparse(cdp_url).netloc netloc = urlparse(cdp_url).netloc
if not netloc: if not netloc: # pragma: no cover
return "Invalid hostname for the CDP URL" return "Invalid hostname for the CDP URL"
return False return False
@@ -89,7 +91,7 @@ class PlaywrightConfig(Struct, kw_only=True, frozen=False, weakref=True):
selector_config: Optional[Dict] = {} selector_config: Optional[Dict] = {}
additional_args: Optional[Dict] = {} additional_args: Optional[Dict] = {}
def __post_init__(self): def __post_init__(self): # pragma: no cover
"""Custom validation after msgspec validation""" """Custom validation after msgspec validation"""
if self.page_action and not callable(self.page_action): if self.page_action and not callable(self.page_action):
raise TypeError(f"page_action must be callable, got {type(self.page_action).__name__}") raise TypeError(f"page_action must be callable, got {type(self.page_action).__name__}")
@@ -194,16 +196,24 @@ class _fetch_params:
def validate_fetch( def validate_fetch(
params: List[Tuple], model: type[PlaywrightConfig] | type[CamoufoxConfig], sentinel=None method_kwargs: Dict | PlaywrightFetchParams | CamoufoxFetchParams,
) -> _fetch_params: session: Any,
model: type[PlaywrightConfig] | type[CamoufoxConfig],
) -> _fetch_params: # pragma: no cover
result = {} result = {}
overrides = {} overrides = {}
for arg, request_value, session_value in params: # Get all field names that _fetch_params needs
if request_value is not sentinel: fetch_param_fields = {f.name for f in fields(_fetch_params)}
overrides[arg] = request_value
for key in fetch_param_fields:
if key in method_kwargs:
overrides[key] = method_kwargs[key]
else: else:
result[arg] = session_value # Check for underscore-prefixed attribute (private)
attr_name = f"_{key}"
if hasattr(session, attr_name):
result[key] = getattr(session, attr_name)
if overrides: if overrides:
validated_config = validate(overrides, model) validated_config = validate(overrides, model)
@@ -213,12 +223,12 @@ def validate_fetch(
for f in fields(_fetch_params) for f in fields(_fetch_params)
if hasattr(validated_config, f.name) if hasattr(validated_config, f.name)
} }
# solve_cloudflare defaults to False for models that don't have it (PlaywrightConfig)
validated_dict.setdefault("solve_cloudflare", False) validated_dict.setdefault("solve_cloudflare", False)
validated_dict.update(result) # Start with session defaults, then overwrite with validated overrides
return _fetch_params(**validated_dict) result.update(validated_dict)
# solve_cloudflare defaults to False for models that don't have it (PlaywrightConfig)
result.setdefault("solve_cloudflare", False) result.setdefault("solve_cloudflare", False)
return _fetch_params(**result) return _fetch_params(**result)
File diff suppressed because it is too large Load Diff
+70 -176
View File
@@ -1,10 +1,5 @@
from scrapling.core._types import ( from scrapling.core._types import Unpack
Callable, from scrapling.engines._browsers._types import PlaywrightSession
List,
Dict,
Optional,
SelectorWaitStates,
)
from scrapling.engines.toolbelt.custom import BaseFetcher, Response from scrapling.engines.toolbelt.custom import BaseFetcher, Response
from scrapling.engines._browsers._controllers import DynamicSession, AsyncDynamicSession from scrapling.engines._browsers._controllers import DynamicSession, AsyncDynamicSession
@@ -26,190 +21,89 @@ class DynamicFetcher(BaseFetcher):
""" """
@classmethod @classmethod
def fetch( def fetch(cls, url: str, **kwargs: Unpack[PlaywrightSession]) -> Response:
cls,
url: str,
headless: bool = True,
google_search: bool = True,
hide_canvas: bool = False,
disable_webgl: bool = False,
real_chrome: bool = False,
stealth: bool = False,
wait: int | float = 0,
page_action: Optional[Callable] = None,
proxy: Optional[str | Dict[str, str]] = None,
locale: str = "en-US",
extra_headers: Optional[Dict[str, str]] = None,
useragent: Optional[str] = None,
cdp_url: Optional[str] = None,
timeout: int | float = 30000,
disable_resources: bool = False,
wait_selector: Optional[str] = None,
init_script: Optional[str] = None,
cookies: Optional[List[Dict]] = None,
network_idle: bool = False,
load_dom: bool = True,
wait_selector_state: SelectorWaitStates = "attached",
extra_flags: Optional[List[str]] = None,
additional_args: Optional[Dict] = None,
custom_config: Optional[Dict] = None,
) -> Response:
"""Opens up a browser and do your request based on your chosen options below. """Opens up a browser and do your request based on your chosen options below.
:param url: Target url. :param url: Target url.
:param headless: Run the browser in headless/hidden (default), or headful/visible mode. :param kwargs: Browser session configuration options including:
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - headless: Run the browser in headless/hidden (default), or headful/visible mode.
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. - disable_resources: Drop requests of unnecessary resources for a speed boost.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. - useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it.
:param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it. - cookies: Set cookies for the next request.
:param cookies: Set cookies for the next request. - network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute.
:param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the Response object.
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - page_action: Added for automation. A function that takes the `page` object and does the automation you need.
:param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - wait_selector: Wait for a specific CSS selector to be in a specific state.
:param wait_selector: Wait for a specific CSS selector to be in a specific state. - init_script: An absolute path to a JavaScript file to be executed on page creation with this request.
:param init_script: An absolute path to a JavaScript file to be executed on page creation with this request. - locale: Set the locale for the browser if wanted. The default value is `en-US`.
:param locale: Set the locale for the browser if wanted. The default value is `en-US`. - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - stealth: Enables stealth mode, check the documentation to see what stealth mode does currently.
:param stealth: Enables stealth mode, check the documentation to see what stealth mode does currently. - real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it.
:param real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it. - hide_canvas: Add random noise to canvas operations to prevent fingerprinting.
:param hide_canvas: Add random noise to canvas operations to prevent fingerprinting. - disable_webgl: Disables WebGL and WebGL 2.0 support entirely.
:param disable_webgl: Disables WebGL and WebGL 2.0 support entirely. - cdp_url: Instead of launching a new browser instance, connect to this CDP URL to control real browsers through CDP.
:param cdp_url: Instead of launching a new browser instance, connect to this CDP URL to control real browsers through CDP. - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. - extra_headers: A dictionary of extra headers to add to the request.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. - extra_flags: A list of additional browser flags to pass to the browser on launch.
:param extra_flags: A list of additional browser flags to pass to the browser on launch. - selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values. - additional_args: Additional arguments to be passed to Playwright's context as additional settings.
:param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings.
:return: A `Response` object. :return: A `Response` object.
""" """
if not custom_config: selector_config = kwargs.get("selector_config", {}) or kwargs.get(
custom_config = {} "custom_config", {}
elif not isinstance(custom_config, dict): ) # Checking `custom_config` for backward compatibility
raise ValueError(f"The custom parser config must be of type dictionary, got {cls.__class__}") if not isinstance(selector_config, dict):
raise TypeError("Argument `selector_config` must be a dictionary.")
with DynamicSession( kwargs["selector_config"] = {**cls._generate_parser_arguments(), **selector_config}
wait=wait,
proxy=proxy, with DynamicSession(**kwargs) as session:
locale=locale,
timeout=timeout,
stealth=stealth,
cdp_url=cdp_url,
cookies=cookies,
headless=headless,
load_dom=load_dom,
useragent=useragent,
real_chrome=real_chrome,
page_action=page_action,
hide_canvas=hide_canvas,
init_script=init_script,
network_idle=network_idle,
google_search=google_search,
extra_headers=extra_headers,
wait_selector=wait_selector,
disable_webgl=disable_webgl,
extra_flags=extra_flags,
additional_args=additional_args,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
selector_config={**cls._generate_parser_arguments(), **custom_config},
) as session:
return session.fetch(url) return session.fetch(url)
@classmethod @classmethod
async def async_fetch( async def async_fetch(cls, url: str, **kwargs: Unpack[PlaywrightSession]) -> Response:
cls,
url: str,
headless: bool = True,
google_search: bool = True,
hide_canvas: bool = False,
disable_webgl: bool = False,
real_chrome: bool = False,
stealth: bool = False,
wait: int | float = 0,
page_action: Optional[Callable] = None,
proxy: Optional[str | Dict[str, str]] = None,
locale: str = "en-US",
extra_headers: Optional[Dict[str, str]] = None,
useragent: Optional[str] = None,
cdp_url: Optional[str] = None,
timeout: int | float = 30000,
disable_resources: bool = False,
wait_selector: Optional[str] = None,
init_script: Optional[str] = None,
cookies: Optional[List[Dict]] = None,
network_idle: bool = False,
load_dom: bool = True,
wait_selector_state: SelectorWaitStates = "attached",
extra_flags: Optional[List[str]] = None,
additional_args: Optional[Dict] = None,
custom_config: Optional[Dict] = None,
) -> Response:
"""Opens up a browser and do your request based on your chosen options below. """Opens up a browser and do your request based on your chosen options below.
:param url: Target url. :param url: Target url.
:param headless: Run the browser in headless/hidden (default), or headful/visible mode. :param kwargs: Browser session configuration options including:
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - headless: Run the browser in headless/hidden (default), or headful/visible mode.
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. - disable_resources: Drop requests of unnecessary resources for a speed boost.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. - useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it.
:param useragent: Pass a useragent string to be used. Otherwise the fetcher will generate a real Useragent of the same browser and use it. - cookies: Set cookies for the next request.
:param cookies: Set cookies for the next request. - network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute.
:param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the Response object.
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - page_action: Added for automation. A function that takes the `page` object and does the automation you need.
:param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - wait_selector: Wait for a specific CSS selector to be in a specific state.
:param wait_selector: Wait for a specific CSS selector to be in a specific state. - init_script: An absolute path to a JavaScript file to be executed on page creation with this request.
:param init_script: An absolute path to a JavaScript file to be executed on page creation with this request. - locale: Set the locale for the browser if wanted. The default value is `en-US`.
:param locale: Set the locale for the browser if wanted. The default value is `en-US`. - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - stealth: Enables stealth mode, check the documentation to see what stealth mode does currently.
:param stealth: Enables stealth mode, check the documentation to see what stealth mode does currently. - real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it.
:param real_chrome: If you have a Chrome browser installed on your device, enable this, and the Fetcher will launch an instance of your browser and use it. - hide_canvas: Add random noise to canvas operations to prevent fingerprinting.
:param hide_canvas: Add random noise to canvas operations to prevent fingerprinting. - disable_webgl: Disables WebGL and WebGL 2.0 support entirely.
:param disable_webgl: Disables WebGL and WebGL 2.0 support entirely. - cdp_url: Instead of launching a new browser instance, connect to this CDP URL to control real browsers through CDP.
:param cdp_url: Instead of launching a new browser instance, connect to this CDP URL to control real browsers through CDP. - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. - extra_headers: A dictionary of extra headers to add to the request.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only. - extra_flags: A list of additional browser flags to pass to the browser on launch.
:param extra_flags: A list of additional browser flags to pass to the browser on launch. - selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values. - additional_args: Additional arguments to be passed to Playwright's context as additional settings.
:param additional_args: Additional arguments to be passed to Playwright's context as additional settings, and it takes higher priority than Scrapling's settings.
:return: A `Response` object. :return: A `Response` object.
""" """
if not custom_config: selector_config = kwargs.get("selector_config", {}) or kwargs.get(
custom_config = {} "custom_config", {}
elif not isinstance(custom_config, dict): ) # Checking `custom_config` for backward compatibility
raise ValueError(f"The custom parser config must be of type dictionary, got {cls.__class__}") if not isinstance(selector_config, dict):
raise TypeError("Argument `selector_config` must be a dictionary.")
async with AsyncDynamicSession( kwargs["selector_config"] = {**cls._generate_parser_arguments(), **selector_config}
wait=wait,
max_pages=1, async with AsyncDynamicSession(**kwargs) as session:
proxy=proxy,
locale=locale,
timeout=timeout,
stealth=stealth,
cdp_url=cdp_url,
cookies=cookies,
headless=headless,
load_dom=load_dom,
useragent=useragent,
real_chrome=real_chrome,
page_action=page_action,
hide_canvas=hide_canvas,
init_script=init_script,
network_idle=network_idle,
google_search=google_search,
extra_headers=extra_headers,
wait_selector=wait_selector,
disable_webgl=disable_webgl,
extra_flags=extra_flags,
additional_args=additional_args,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
selector_config={**cls._generate_parser_arguments(), **custom_config},
) as session:
return await session.fetch(url) return await session.fetch(url)
+72 -182
View File
@@ -1,10 +1,5 @@
from scrapling.core._types import ( from scrapling.core._types import Unpack
Callable, from scrapling.engines._browsers._types import CamoufoxSession
Dict,
List,
Optional,
SelectorWaitStates,
)
from scrapling.engines.toolbelt.custom import BaseFetcher, Response from scrapling.engines.toolbelt.custom import BaseFetcher, Response
from scrapling.engines._browsers._camoufox import StealthySession, AsyncStealthySession from scrapling.engines._browsers._camoufox import StealthySession, AsyncStealthySession
@@ -17,196 +12,91 @@ class StealthyFetcher(BaseFetcher):
""" """
@classmethod @classmethod
def fetch( def fetch(cls, url: str, **kwargs: Unpack[CamoufoxSession]) -> Response:
cls,
url: str,
headless: bool = True, # noqa: F821
block_images: bool = False,
disable_resources: bool = False,
block_webrtc: bool = False,
allow_webgl: bool = True,
network_idle: bool = False,
load_dom: bool = True,
humanize: bool | float = True,
solve_cloudflare: bool = False,
wait: int | float = 0,
timeout: int | float = 30000,
page_action: Optional[Callable] = None,
wait_selector: Optional[str] = None,
init_script: Optional[str] = None,
addons: Optional[List[str]] = None,
wait_selector_state: SelectorWaitStates = "attached",
cookies: Optional[List[Dict]] = None,
google_search: bool = True,
extra_headers: Optional[Dict[str, str]] = None,
proxy: Optional[str | Dict[str, str]] = None,
os_randomize: bool = False,
disable_ads: bool = False,
geoip: bool = False,
custom_config: Optional[Dict] = None,
additional_args: Optional[Dict] = None,
) -> Response:
""" """
Opens up a browser and do your request based on your chosen options below. Opens up a browser and do your request based on your chosen options below.
:param url: Target url. :param url: Target url.
:param headless: Run the browser in headless/hidden (default), or headful/visible mode. :param kwargs: Browser session configuration options including:
:param block_images: Prevent the loading of images through Firefox preferences. - headless: Run the browser in headless/hidden (default), or headful/visible mode.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. - block_images: Prevent the loading of images through Firefox preferences.
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - disable_resources: Drop requests of unnecessary resources for a speed boost.
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. - block_webrtc: Blocks WebRTC entirely.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. - allow_webgl: Enabled by default. Disabling WebGL is not recommended as many WAFs now check if WebGL is enabled.
:param block_webrtc: Blocks WebRTC entirely. - network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param cookies: Set cookies for the next request. - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute.
:param addons: List of Firefox addons to use. Must be paths to extracted addons. - humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement.
:param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window. - solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you.
:param solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you. - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the Response object.
:param allow_webgl: Enabled by default. Disabling WebGL is not recommended as many WAFs now check if WebGL is enabled. - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. - page_action: Added for automation. A function that takes the `page` object and does the automation you need.
:param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - wait_selector: Wait for a specific CSS selector to be in a specific state.
:param disable_ads: Disabled by default, this installs the `uBlock Origin` addon on the browser if enabled. - init_script: An absolute path to a JavaScript file to be executed on page creation with this request.
:param os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. The default is Scrapling matching the fingerprints with the current OS. - addons: List of Firefox addons to use. Must be paths to extracted addons.
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - cookies: Set cookies for the next request.
:param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param wait_selector: Wait for a specific CSS selector to be in a specific state. - extra_headers: A dictionary of extra headers to add to the request.
:param init_script: An absolute path to a JavaScript file to be executed on page creation with this request. - proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
:param geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, and spoof the WebRTC IP address. - os_randomize: If enabled, Scrapling will randomize the OS fingerprints used.
It will also calculate and spoof the browser's language based on the distribution of language speakers in the target region. - disable_ads: Disabled by default, this installs the `uBlock Origin` addon on the browser if enabled.
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, and spoof the WebRTC IP address.
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. - selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - additional_args: Additional arguments to be passed to Camoufox as additional settings.
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
:param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values.
:param additional_args: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings.
:return: A `Response` object. :return: A `Response` object.
""" """
if not custom_config: selector_config = kwargs.get("selector_config", {}) or kwargs.get(
custom_config = {} "custom_config", {}
) # Checking `custom_config` for backward compatibility
if not isinstance(selector_config, dict):
raise TypeError("Argument `selector_config` must be a dictionary.")
with StealthySession( kwargs["selector_config"] = {**cls._generate_parser_arguments(), **selector_config}
wait=wait,
proxy=proxy, with StealthySession(**kwargs) as engine:
geoip=geoip,
addons=addons,
timeout=timeout,
cookies=cookies,
headless=headless,
humanize=humanize,
load_dom=load_dom,
disable_ads=disable_ads,
allow_webgl=allow_webgl,
page_action=page_action,
init_script=init_script,
network_idle=network_idle,
block_images=block_images,
block_webrtc=block_webrtc,
os_randomize=os_randomize,
wait_selector=wait_selector,
google_search=google_search,
extra_headers=extra_headers,
solve_cloudflare=solve_cloudflare,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
selector_config={**cls._generate_parser_arguments(), **custom_config},
additional_args=additional_args or {},
) as engine:
return engine.fetch(url) return engine.fetch(url)
@classmethod @classmethod
async def async_fetch( async def async_fetch(cls, url: str, **kwargs: Unpack[CamoufoxSession]) -> Response:
cls,
url: str,
headless: bool = True, # noqa: F821
block_images: bool = False,
disable_resources: bool = False,
block_webrtc: bool = False,
allow_webgl: bool = True,
network_idle: bool = False,
load_dom: bool = True,
humanize: bool | float = True,
solve_cloudflare: bool = False,
wait: int | float = 0,
timeout: int | float = 30000,
page_action: Optional[Callable] = None,
wait_selector: Optional[str] = None,
init_script: Optional[str] = None,
addons: Optional[List[str]] = None,
wait_selector_state: SelectorWaitStates = "attached",
cookies: Optional[List[Dict]] = None,
google_search: bool = True,
extra_headers: Optional[Dict[str, str]] = None,
proxy: Optional[str | Dict[str, str]] = None,
os_randomize: bool = False,
disable_ads: bool = False,
geoip: bool = False,
custom_config: Optional[Dict] = None,
additional_args: Optional[Dict] = None,
) -> Response:
""" """
Opens up a browser and do your request based on your chosen options below. Opens up a browser and do your request based on your chosen options below.
:param url: Target url. :param url: Target url.
:param headless: Run the browser in headless/hidden (default), or headful/visible mode. :param kwargs: Browser session configuration options including:
:param block_images: Prevent the loading of images through Firefox preferences. - headless: Run the browser in headless/hidden (default), or headful/visible mode.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. - block_images: Prevent the loading of images through Firefox preferences.
:param disable_resources: Drop requests of unnecessary resources for a speed boost. It depends, but it made requests ~25% faster in my tests for some websites. - disable_resources: Drop requests of unnecessary resources for a speed boost.
Requests dropped are of type `font`, `image`, `media`, `beacon`, `object`, `imageset`, `texttrack`, `websocket`, `csp_report`, and `stylesheet`. - block_webrtc: Blocks WebRTC entirely.
This can help save your proxy usage but be careful with this option as it makes some websites never finish loading. - allow_webgl: Enabled by default. Disabling WebGL is not recommended as many WAFs now check if WebGL is enabled.
:param block_webrtc: Blocks WebRTC entirely. - network_idle: Wait for the page until there are no network connections for at least 500 ms.
:param cookies: Set cookies for the next request. - load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute.
:param addons: List of Firefox addons to use. Must be paths to extracted addons. - humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement.
:param humanize: Humanize the cursor movement. Takes either True or the MAX duration in seconds of the cursor movement. The cursor typically takes up to 1.5 seconds to move across the window. - solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you.
:param solve_cloudflare: Solves all types of the Cloudflare's Turnstile/Interstitial challenges before returning the response to you. - wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the Response object.
:param allow_webgl: Enabled by default. Disabling WebGL is not recommended as many WAFs now check if WebGL is enabled. - timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000
:param network_idle: Wait for the page until there are no network connections for at least 500 ms. - page_action: Added for automation. A function that takes the `page` object and does the automation you need.
:param load_dom: Enabled by default, wait for all JavaScript on page(s) to fully load and execute. - wait_selector: Wait for a specific CSS selector to be in a specific state.
:param disable_ads: Disabled by default, this installs the `uBlock Origin` addon on the browser if enabled. - init_script: An absolute path to a JavaScript file to be executed on page creation with this request.
:param os_randomize: If enabled, Scrapling will randomize the OS fingerprints used. The default is Scrapling matching the fingerprints with the current OS. - addons: List of Firefox addons to use. Must be paths to extracted addons.
:param wait: The time (milliseconds) the fetcher will wait after everything finishes before closing the page and returning the ` Response ` object. - wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`.
:param timeout: The timeout in milliseconds that is used in all operations and waits through the page. The default is 30,000 - cookies: Set cookies for the next request.
:param page_action: Added for automation. A function that takes the `page` object and does the automation you need. - google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name.
:param wait_selector: Wait for a specific CSS selector to be in a specific state. - extra_headers: A dictionary of extra headers to add to the request.
:param init_script: An absolute path to a JavaScript file to be executed on page creation with this request. - proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
:param geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, and spoof the WebRTC IP address. - os_randomize: If enabled, Scrapling will randomize the OS fingerprints used.
It will also calculate and spoof the browser's language based on the distribution of language speakers in the target region. - disable_ads: Disabled by default, this installs the `uBlock Origin` addon on the browser if enabled.
:param wait_selector_state: The state to wait for the selector given with `wait_selector`. The default state is `attached`. - geoip: Recommended to use with proxies; Automatically use IP's longitude, latitude, timezone, country, locale, and spoof the WebRTC IP address.
:param google_search: Enabled by default, Scrapling will set the referer header to be as if this request came from a Google search of this website's domain name. - selector_config: The arguments that will be passed in the end while creating the final Selector's class.
:param extra_headers: A dictionary of extra headers to add to the request. _The referer set by the `google_search` argument takes priority over the referer set here if used together._ - additional_args: Additional arguments to be passed to Camoufox as additional settings.
:param proxy: The proxy to be used with requests, it can be a string or a dictionary with the keys 'server', 'username', and 'password' only.
:param custom_config: A dictionary of custom parser arguments to use with this request. Any argument passed will override any class parameters values.
:param additional_args: Additional arguments to be passed to Camoufox as additional settings, and it takes higher priority than Scrapling's settings.
:return: A `Response` object. :return: A `Response` object.
""" """
if not custom_config: selector_config = kwargs.get("selector_config", {}) or kwargs.get(
custom_config = {} "custom_config", {}
) # Checking `custom_config` for backward compatibility
if not isinstance(selector_config, dict):
raise TypeError("Argument `selector_config` must be a dictionary.")
async with AsyncStealthySession( kwargs["selector_config"] = {**cls._generate_parser_arguments(), **selector_config}
wait=wait,
max_pages=1, async with AsyncStealthySession(**kwargs) as engine:
proxy=proxy,
geoip=geoip,
addons=addons,
timeout=timeout,
cookies=cookies,
headless=headless,
humanize=humanize,
load_dom=load_dom,
disable_ads=disable_ads,
allow_webgl=allow_webgl,
page_action=page_action,
init_script=init_script,
network_idle=network_idle,
block_images=block_images,
block_webrtc=block_webrtc,
os_randomize=os_randomize,
wait_selector=wait_selector,
google_search=google_search,
extra_headers=extra_headers,
solve_cloudflare=solve_cloudflare,
disable_resources=disable_resources,
wait_selector_state=wait_selector_state,
selector_config={**cls._generate_parser_arguments(), **custom_config},
additional_args=additional_args or {},
) as engine:
return await engine.fetch(url) return await engine.fetch(url)
+1 -1
View File
@@ -121,7 +121,7 @@ class Selector(SelectorsGeneration):
self.__text = None self.__text = None
if root is None: if root is None:
if isinstance(content, str): if isinstance(content, str):
body = content.strip().replace("\x00", "").encode(encoding) or b"<html/>" body = content.strip().replace("\x00", "") or "<html/>"
elif isinstance(content, bytes): elif isinstance(content, bytes):
body = content.replace(b"\x00", b"") body = content.replace(b"\x00", b"")
else: else:
+1 -1
View File
@@ -1,6 +1,6 @@
[metadata] [metadata]
name = scrapling name = scrapling
version = 0.3.9 version = 0.3.10
author = Karim Shoair author = Karim Shoair
author_email = karim.shoair@pm.me author_email = karim.shoair@pm.me
description = Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be! description = Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!
+1 -1
View File
@@ -70,7 +70,7 @@ class TestStealthyFetcher:
"os_randomize": True, "os_randomize": True,
"disable_ads": True, "disable_ads": True,
# "geoip": True, # "geoip": True,
"custom_config": {"keep_comments": False, "keep_cdata": False}, "selector_config": {"keep_comments": False, "keep_cdata": False},
"additional_args": {"window": (1920, 1080)}, "additional_args": {"window": (1920, 1080)},
}, },
], ],
+1 -1
View File
@@ -68,7 +68,7 @@ class TestDynamicFetcherAsync:
"useragent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:131.0) Gecko/20100101 Firefox/131.0", "useragent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:131.0) Gecko/20100101 Firefox/131.0",
"cookies": [{"name": "test", "value": "123", "domain": "example.com", "path": "/"}], "cookies": [{"name": "test", "value": "123", "domain": "example.com", "path": "/"}],
"network_idle": True, "network_idle": True,
"custom_config": {"keep_comments": False, "keep_cdata": False}, "selector_config": {"keep_comments": False, "keep_cdata": False},
}, },
], ],
) )
+1 -1
View File
@@ -65,7 +65,7 @@ class TestStealthyFetcher:
"os_randomize": True, "os_randomize": True,
"disable_ads": True, "disable_ads": True,
# "geoip": True, # "geoip": True,
"custom_config": {"keep_comments": False, "keep_cdata": False}, "selector_config": {"keep_comments": False, "keep_cdata": False},
"additional_args": {"window": (1920, 1080)}, "additional_args": {"window": (1920, 1080)},
}, },
], ],
+6 -6
View File
@@ -63,12 +63,12 @@ class TestStealthySession:
) as session: ) as session:
assert session.max_pages == 1 assert session.max_pages == 1
assert session.headless is True assert session._headless is True
assert session.block_images is True assert session._block_images is True
assert session.disable_resources is True assert session._disable_resources is True
assert session.solve_cloudflare is True assert session._solve_cloudflare is True
assert session.wait == 1000 assert session._wait == 1000
assert session.timeout == 60000 assert session._timeout == 60000
assert session.context is not None assert session.context is not None
# Test Cloudflare detection # Test Cloudflare detection
+1 -1
View File
@@ -66,7 +66,7 @@ class TestDynamicFetcher:
"useragent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:131.0) Gecko/20100101 Firefox/131.0", "useragent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:131.0) Gecko/20100101 Firefox/131.0",
"cookies": [{"name": "test", "value": "123", "domain": "example.com", "path": "/"}], "cookies": [{"name": "test", "value": "123", "domain": "example.com", "path": "/"}],
"network_idle": True, "network_idle": True,
"custom_config": {"keep_comments": False, "keep_cdata": False}, "selector_config": {"keep_comments": False, "keep_cdata": False},
}, },
], ],
) )