Compare commits
33 Commits
6390c0af3d
...
6d2c9d95b9
| Author | SHA1 | Date | |
|---|---|---|---|
| 6d2c9d95b9 | |||
| d3046dd00a | |||
| ea53dced9d | |||
| c0f012d662 | |||
| 602f72ea60 | |||
| 149050685e | |||
| 74c5848060 | |||
| 5fdb46fea0 | |||
| 061f778434 | |||
| 9352ccec3f | |||
| b3408c6f3c | |||
| c732c4ef31 | |||
| c84764d627 | |||
| 4ced6e4ac2 | |||
| 1fd1013806 | |||
| 3d96baf284 | |||
| 9c0c857245 | |||
| f7d1e338e0 | |||
| 4f0a593b7d | |||
| 2b2abb3810 | |||
| 507f7f626d | |||
| 5d839c5995 | |||
| fa5f1477d0 | |||
| b9bc17b461 | |||
| f0db3d7d14 | |||
| 141b389cdb | |||
| cd4cdc69e6 | |||
| 3a2dd29255 | |||
| 53ef32c723 | |||
| 6efd50dd17 | |||
| dbc6817f73 | |||
| 9e43062a63 | |||
| c243c8aca5 |
@@ -73,7 +73,7 @@ jobs:
|
||||
- name: Install all browsers dependencies
|
||||
run: |
|
||||
python3 -m pip install --upgrade pip
|
||||
python3 -m pip install playwright==1.59.0 patchright==1.59.1
|
||||
python3 -m pip install playwright==1.60.0 patchright==1.60.1
|
||||
|
||||
- name: Get Playwright version
|
||||
id: playwright-version
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@ There are many ways to contribute to Scrapling. Here are some of them:
|
||||
- Report bugs and request features using the [GitHub issues](https://github.com/D4Vinci/Scrapling/issues). Please follow the issue template to help us resolve your issue quickly.
|
||||
- Blog about Scrapling. Tell the world how you’re using Scrapling. This will help newcomers with more examples and increase the Scrapling project's visibility.
|
||||
- Join the [Discord community](https://discord.gg/EMgGbDceNQ) and share your ideas on how to improve Scrapling. We’re always open to suggestions.
|
||||
- If you are not a developer, perhaps you would like to help with translating the [documentation](https://github.com/D4Vinci/Scrapling/tree/docs)?
|
||||
- If you are not a developer, perhaps you would like to help with translating the [documentation](https://github.com/D4Vinci/Scrapling/tree/dev/docs)?
|
||||
|
||||
## Making a Pull Request
|
||||
To ensure that your PR gets accepted, please make sure that your PR is based on the latest changes from the dev branch and that it satisfies the following requirements:
|
||||
|
||||
@@ -87,6 +87,15 @@ MySpider().start()
|
||||
|
||||
# Platinum Sponsors
|
||||
<table>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank" title="Clean Proxies with No Nonsense.">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxidize.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank"><b>Proxidize</b></a> provides mobile and residential proxies for scraping, browser automation, SEO monitoring, AI agents, and data collection. <i>Use code <b>scrapling20</b> for 20% off</i>.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
@@ -141,16 +150,6 @@ MySpider().start()
|
||||
<a href="https://tikhub.io/?utm_source=github.com/D4Vinci/Scrapling&utm_medium=marketing_social&utm_campaign=retargeting&utm_content=carousel_ad" target="_blank">TikHub.io</a> provides 900+ stable APIs across 16+ platforms including TikTok, X, YouTube & Instagram, with 40M+ datasets. <br /> Also offers <a href="https://ai.tikhub.io/?ref=KarimShoair" target="_blank">DISCOUNTED AI models</a> - Claude, GPT, GEMINI & more up to 71% off.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank" title="Scalable Web Data Access for AI Applications">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/nsocks.png">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank">Nsocks</a> provides fast Residential and ISP proxies for developers and scrapers. Global IP coverage, high anonymity, smart rotation, and reliable performance for automation and data extraction. Use <a href="https://www.xcrawl.com/?keyword=2p67aivg" target="_blank">Xcrawl</a> to simplify large-scale web crawling.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://petrosky.io/d4vinci" target="_blank" title="PetroSky delivers cutting-edge VPS hosting.">
|
||||
@@ -189,17 +188,17 @@ MySpider().start()
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> provides residential proxies from just $0.015/IP or $0.68/GB. 20M+ IPs across 90+ countries. Sticky or rotating sessions, managed from desktop or mobile app.
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> provides residential proxies from just $0.018/IP or $0.68/GB. 20M+ IPs across 90+ countries. Sticky or rotating sessions, managed from desktop or mobile app.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.png">
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.svg" width="240" height="100">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank">NodeMaven</a> - reliable proxy provider with the highest quality IP on the market. Use promo code SCRAPLING35 for 35% discount on proxies.
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank">NodeMaven</a> - reliable proxy provider with the highest quality IP on the market. Use promo code SCRAPLING35 for 35% discount on proxies.
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
@@ -215,11 +214,12 @@ MySpider().start()
|
||||
<a href="https://proxyempire.io/?ref=scrapling&utm_source=scrapling" target="_blank" title="Collect The Data Your Project Needs with the Best Residential Proxies"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/ProxyEmpire.png"></a>
|
||||
<a href="https://www.webshare.io/?referral_code=48r2m2cd5uz1" target="_blank" title="The Most Reliable Proxy with Unparalleled Performance"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/webshare.png"></a>
|
||||
<a href="https://proxiware.com/?ref=scrapling" target="_blank" title="Collect Any Data. At Any Scale."><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxiware.png"></a>
|
||||
<a href="https://talordata.com/?campaignid=ZOYRsA4bX9BwWyO5&utm_source=D4Vinci&utm_term=Scrapling" target="_blank" title="Google SERP API for LLM, AI Agents & Full-Stack SEO"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/TalorData.jpg"></a>
|
||||
|
||||
|
||||
<!-- /sponsors -->
|
||||
|
||||
<i><sub>Do you want to show your ad here? Click [here](https://github.com/sponsors/D4Vinci) and choose the tier that suites you!</sub></i>
|
||||
<i><sub>Do you want to show your ad here? Click [here](https://github.com/sponsors/D4Vinci) and choose the tier that suits you!</sub></i>
|
||||
|
||||
---
|
||||
|
||||
@@ -481,7 +481,8 @@ Scrapling requires Python 3.10 or higher:
|
||||
pip install scrapling
|
||||
```
|
||||
|
||||
This installation only includes the parser engine and its dependencies, without any fetchers or commandline dependencies.
|
||||
> [!IMPORTANT]
|
||||
> This installation only includes the parser engine and its dependencies, without any fetchers or commandline dependencies. So importing anything from `scrapling.fetchers` or `scrapling.spiders`, like in the examples above, will raise `ModuleNotFoundError` with this installation alone. If you are going to use any of the fetchers or spiders, install the fetchers' dependencies first as shown below.
|
||||
|
||||
### Optional Dependencies
|
||||
|
||||
|
||||
Binary file not shown.
@@ -1,7 +1,7 @@
|
||||
---
|
||||
name: scrapling-official
|
||||
description: Scrape web pages using Scrapling with anti-bot bypass (like Cloudflare Turnstile), stealth headless browsing, spiders framework, adaptive scraping, and JavaScript rendering. Use when asked to scrape, crawl, or extract data from websites; web_fetch fails; the site has anti-bot protections; write Python code to scrape/crawl; or write spiders.
|
||||
version: "0.4.8"
|
||||
version: "0.4.9"
|
||||
license: Complete terms in LICENSE.txt
|
||||
metadata:
|
||||
homepage: "https://scrapling.readthedocs.io/en/latest/index.html"
|
||||
@@ -40,7 +40,7 @@ Blazing fast crawls with real-time stats and streaming. Built by Web Scrapers fo
|
||||
|
||||
Create a virtual Python environment through any way available, like `venv`, then inside the environment do:
|
||||
|
||||
`pip install "scrapling[all]>=0.4.8"`
|
||||
`pip install "scrapling[all]>=0.4.9"`
|
||||
|
||||
Then do this to download all the browsers' dependencies:
|
||||
|
||||
|
||||
@@ -9,7 +9,7 @@ All examples collect **all 100 quotes across 10 pages**.
|
||||
Make sure Scrapling is installed:
|
||||
|
||||
```bash
|
||||
pip install "scrapling[all]>=0.4.8"
|
||||
pip install "scrapling[all]>=0.4.9"
|
||||
scrapling install --force
|
||||
```
|
||||
|
||||
|
||||
@@ -208,7 +208,7 @@ Docker alternative:
|
||||
|
||||
```bash
|
||||
docker pull pyd4vinci/scrapling
|
||||
docker run -i --rm scrapling mcp
|
||||
docker run -i --rm pyd4vinci/scrapling mcp
|
||||
```
|
||||
|
||||
The MCP server name when registering with a client is `ScraplingServer`. The command is the path to the `scrapling` binary and the argument is `mcp`.
|
||||
+16
-15
@@ -83,6 +83,15 @@ MySpider().start()
|
||||
|
||||
# الرعاة البلاتينيون
|
||||
<table>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank" title="Clean Proxies with No Nonsense.">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxidize.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> توفر <a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank"><b>Proxidize</b></a> وكلاء جوّالين وسكنيين للاستخراج، وأتمتة المتصفح، ومراقبة تحسين محركات البحث (SEO)، ووكلاء الذكاء الاصطناعي، وجمع البيانات. <i>استخدم الرمز <b>scrapling20</b> للحصول على خصم 20%</i>.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
@@ -137,16 +146,6 @@ MySpider().start()
|
||||
<a href="https://tikhub.io/?utm_source=github.com/D4Vinci/Scrapling&utm_medium=marketing_social&utm_campaign=retargeting&utm_content=carousel_ad" target="_blank">TikHub.io</a> يوفر أكثر من 900 واجهة API مستقرة عبر أكثر من 16 منصة تشمل TikTok و X و YouTube و Instagram، مع أكثر من 40 مليون مجموعة بيانات. <br /> يقدم أيضاً <a href="https://ai.tikhub.io/?ref=KarimShoair" target="_blank">نماذج ذكاء اصطناعي بأسعار مخفضة</a> - Claude و GPT و GEMINI والمزيد بخصم يصل إلى 71%.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank" title="Scalable Web Data Access for AI Applications">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/nsocks.png">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank">Nsocks</a> يوفر بروكسيات سكنية و ISP سريعة للمطورين والسكرابرز. تغطية IP عالمية، إخفاء هوية عالي، تدوير ذكي، وأداء موثوق للأتمتة واستخراج البيانات. استخدم <a href="https://www.xcrawl.com/?keyword=2p67aivg" target="_blank">Xcrawl</a> لتبسيط زحف الويب على نطاق واسع.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://petrosky.io/d4vinci" target="_blank" title="PetroSky delivers cutting-edge VPS hosting.">
|
||||
@@ -185,17 +184,17 @@ MySpider().start()
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
يوفر <a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> بروكسيات سكنية بدءًا من 0.015 دولار فقط لكل IP أو 0.68 دولار لكل جيجابايت. أكثر من 20 مليون عنوان IP في أكثر من 90 دولة. جلسات ثابتة أو متناوبة، تتم إدارتها من تطبيق سطح المكتب أو الجوال.
|
||||
يوفر <a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> بروكسيات سكنية بدءًا من 0.018 دولار فقط لكل IP أو 0.68 دولار لكل جيجابايت. أكثر من 20 مليون عنوان IP في أكثر من 90 دولة. جلسات ثابتة أو متناوبة، تتم إدارتها من تطبيق سطح المكتب أو الجوال.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.png">
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.svg" width="240" height="100">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank">NodeMaven</a> - مزود بروكسيات موثوق يقدم أعلى جودة IP في السوق. استخدم كود الخصم SCRAPLING35 للحصول على خصم 35% على البروكسيات.
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank">NodeMaven</a> - مزود بروكسيات موثوق يقدم أعلى جودة IP في السوق. استخدم كود الخصم SCRAPLING35 للحصول على خصم 35% على البروكسيات.
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
@@ -211,6 +210,7 @@ MySpider().start()
|
||||
<a href="https://proxyempire.io/?ref=scrapling&utm_source=scrapling" target="_blank" title="Collect The Data Your Project Needs with the Best Residential Proxies"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/ProxyEmpire.png"></a>
|
||||
<a href="https://www.webshare.io/?referral_code=48r2m2cd5uz1" target="_blank" title="The Most Reliable Proxy with Unparalleled Performance"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/webshare.png"></a>
|
||||
<a href="https://proxiware.com/?ref=scrapling" target="_blank" title="Collect Any Data. At Any Scale."><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxiware.png"></a>
|
||||
<a href="https://talordata.com/?campaignid=ZOYRsA4bX9BwWyO5&utm_source=D4Vinci&utm_term=Scrapling" target="_blank" title="Google SERP API for LLM, AI Agents & Full-Stack SEO"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/TalorData.jpg"></a>
|
||||
|
||||
|
||||
<!-- /sponsors -->
|
||||
@@ -477,7 +477,8 @@ Scrapling ليس قوياً فحسب - بل هو أيضاً سريع بشكل م
|
||||
pip install scrapling
|
||||
```
|
||||
|
||||
يتضمن هذا التثبيت فقط محرك المحلل وتبعياته، بدون أي جوالب أو تبعيات سطر الأوامر.
|
||||
> [!IMPORTANT]
|
||||
> يتضمن هذا التثبيت فقط محرك المحلل وتبعياته، بدون أي جوالب أو تبعيات سطر الأوامر. لذلك، فإن استيراد أي شيء من `scrapling.fetchers` أو `scrapling.spiders`، كما في الأمثلة أعلاه، سيؤدي إلى خطأ `ModuleNotFoundError` مع هذا التثبيت وحده. إذا كنت ستستخدم أيًا من الجوالب أو العناكب، فقم أولًا بتثبيت تبعيات الجوالب كما هو موضح أدناه.
|
||||
|
||||
### التبعيات الاختيارية
|
||||
|
||||
|
||||
+16
-15
@@ -83,6 +83,15 @@ MySpider().start()
|
||||
|
||||
# 铂金赞助商
|
||||
<table>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank" title="Clean Proxies with No Nonsense.">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxidize.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank"><b>Proxidize</b></a> 提供移动代理和住宅代理,适用于网页抓取、浏览器自动化、SEO 监控、AI 代理和数据收集。<i>使用优惠码 <b>scrapling20</b> 可享 20% 折扣</i>。
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
@@ -137,16 +146,6 @@ MySpider().start()
|
||||
<a href="https://tikhub.io/?utm_source=github.com/D4Vinci/Scrapling&utm_medium=marketing_social&utm_campaign=retargeting&utm_content=carousel_ad" target="_blank">TikHub.io</a> 提供覆盖 16+ 平台(包括 TikTok、X、YouTube 和 Instagram)的 900+ 稳定 API,拥有 4000 万+ 数据集。<br /> 还提供<a href="https://ai.tikhub.io/?ref=KarimShoair" target="_blank">优惠 AI 模型</a> - Claude、GPT、GEMINI 等,最高优惠 71%。
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank" title="Scalable Web Data Access for AI Applications">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/nsocks.png">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank">Nsocks</a> 提供面向开发者和爬虫的快速住宅和 ISP 代理。全球 IP 覆盖、高匿名性、智能轮换,以及可靠的自动化和数据提取性能。使用 <a href="https://www.xcrawl.com/?keyword=2p67aivg" target="_blank">Xcrawl</a> 简化大规模网页爬取。
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://petrosky.io/d4vinci" target="_blank" title="PetroSky delivers cutting-edge VPS hosting.">
|
||||
@@ -185,17 +184,17 @@ MySpider().start()
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> 提供住宅代理,价格低至每个 IP 仅 $0.015 或每 GB $0.68。覆盖 90+ 国家/地区的 2000 万+ IP。支持固定或轮换会话,可通过桌面或移动应用进行管理。
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> 提供住宅代理,价格低至每个 IP 仅 $0.018 或每 GB $0.68。覆盖 90+ 国家/地区的 2000 万+ IP。支持固定或轮换会话,可通过桌面或移动应用进行管理。
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.png">
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.svg" width="240" height="100">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank">NodeMaven</a> - 市场上 IP 质量最高的可靠代理提供商。使用优惠码 SCRAPLING35 可享代理 35% 折扣。
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank">NodeMaven</a> - 市场上 IP 质量最高的可靠代理提供商。使用优惠码 SCRAPLING35 可享代理 35% 折扣。
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
@@ -211,6 +210,7 @@ MySpider().start()
|
||||
<a href="https://proxyempire.io/?ref=scrapling&utm_source=scrapling" target="_blank" title="Collect The Data Your Project Needs with the Best Residential Proxies"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/ProxyEmpire.png"></a>
|
||||
<a href="https://www.webshare.io/?referral_code=48r2m2cd5uz1" target="_blank" title="The Most Reliable Proxy with Unparalleled Performance"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/webshare.png"></a>
|
||||
<a href="https://proxiware.com/?ref=scrapling" target="_blank" title="Collect Any Data. At Any Scale."><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxiware.png"></a>
|
||||
<a href="https://talordata.com/?campaignid=ZOYRsA4bX9BwWyO5&utm_source=D4Vinci&utm_term=Scrapling" target="_blank" title="Google SERP API for LLM, AI Agents & Full-Stack SEO"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/TalorData.jpg"></a>
|
||||
|
||||
|
||||
<!-- /sponsors -->
|
||||
@@ -477,7 +477,8 @@ Scrapling 需要 Python 3.10 或更高版本:
|
||||
pip install scrapling
|
||||
```
|
||||
|
||||
此安装仅包括解析器引擎及其依赖项,没有任何 Fetcher 或命令行依赖项。
|
||||
> [!IMPORTANT]
|
||||
> 此安装仅包括解析器引擎及其依赖项,没有任何 Fetcher 或命令行依赖项。 因此,仅使用此安装时,像上面的示例那样从 `scrapling.fetchers` 或 `scrapling.spiders` 导入任何内容都会引发 `ModuleNotFoundError`。如果要使用任何 Fetcher 或 Spider,请先按照下面的说明安装 Fetcher 的依赖项。
|
||||
|
||||
### 可选依赖项
|
||||
|
||||
|
||||
+16
-15
@@ -83,6 +83,15 @@ MySpider().start()
|
||||
|
||||
# Platin-Sponsoren
|
||||
<table>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank" title="Clean Proxies with No Nonsense.">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxidize.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank"><b>Proxidize</b></a> bietet mobile und Residential-Proxies für Scraping, Browser-Automatisierung, SEO-Monitoring, KI-Agenten und Datenerfassung. <i>Mit dem Code <b>scrapling20</b> erhalten Sie 20% Rabatt</i>.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
@@ -137,16 +146,6 @@ MySpider().start()
|
||||
<a href="https://tikhub.io/?utm_source=github.com/D4Vinci/Scrapling&utm_medium=marketing_social&utm_campaign=retargeting&utm_content=carousel_ad" target="_blank">TikHub.io</a> bietet über 900 stabile APIs auf mehr als 16 Plattformen, darunter TikTok, X, YouTube und Instagram, mit über 40 Mio. Datensätzen. <br /> Bietet außerdem <a href="https://ai.tikhub.io/?ref=KarimShoair" target="_blank">vergünstigte KI-Modelle</a> - Claude, GPT, GEMINI und mehr mit bis zu 71% Rabatt.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank" title="Scalable Web Data Access for AI Applications">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/nsocks.png">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank">Nsocks</a> bietet schnelle Residential- und ISP-Proxies für Entwickler und Scraper. Globale IP-Abdeckung, hohe Anonymität, intelligente Rotation und zuverlässige Leistung für Automatisierung und Datenextraktion. Verwenden Sie <a href="https://www.xcrawl.com/?keyword=2p67aivg" target="_blank">Xcrawl</a>, um großflächiges Web-Crawling zu vereinfachen.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://petrosky.io/d4vinci" target="_blank" title="PetroSky delivers cutting-edge VPS hosting.">
|
||||
@@ -185,17 +184,17 @@ MySpider().start()
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> bietet Residential-Proxys ab nur 0,015 $/IP oder 0,68 $/GB. Über 20 Mio. IPs in mehr als 90 Ländern. Sticky oder rotierende Sessions, verwaltet über die Desktop- oder Mobile-App.
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> bietet Residential-Proxys ab nur 0,018 $/IP oder 0,68 $/GB. Über 20 Mio. IPs in mehr als 90 Ländern. Sticky oder rotierende Sessions, verwaltet über die Desktop- oder Mobile-App.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.png">
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.svg" width="240" height="100">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank">NodeMaven</a> - zuverlässiger Proxy-Anbieter mit der höchsten IP-Qualität auf dem Markt. Nutze den Promo-Code SCRAPLING35 für 35% Rabatt auf Proxys.
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank">NodeMaven</a> - zuverlässiger Proxy-Anbieter mit der höchsten IP-Qualität auf dem Markt. Nutze den Promo-Code SCRAPLING35 für 35% Rabatt auf Proxys.
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
@@ -211,6 +210,7 @@ MySpider().start()
|
||||
<a href="https://proxyempire.io/?ref=scrapling&utm_source=scrapling" target="_blank" title="Collect The Data Your Project Needs with the Best Residential Proxies"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/ProxyEmpire.png"></a>
|
||||
<a href="https://www.webshare.io/?referral_code=48r2m2cd5uz1" target="_blank" title="The Most Reliable Proxy with Unparalleled Performance"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/webshare.png"></a>
|
||||
<a href="https://proxiware.com/?ref=scrapling" target="_blank" title="Collect Any Data. At Any Scale."><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxiware.png"></a>
|
||||
<a href="https://talordata.com/?campaignid=ZOYRsA4bX9BwWyO5&utm_source=D4Vinci&utm_term=Scrapling" target="_blank" title="Google SERP API for LLM, AI Agents & Full-Stack SEO"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/TalorData.jpg"></a>
|
||||
|
||||
|
||||
<!-- /sponsors -->
|
||||
@@ -477,7 +477,8 @@ Scrapling erfordert Python 3.10 oder höher:
|
||||
pip install scrapling
|
||||
```
|
||||
|
||||
Diese Installation enthält nur die Parser-Engine und ihre Abhängigkeiten, ohne Fetcher oder Kommandozeilenabhängigkeiten.
|
||||
> [!IMPORTANT]
|
||||
> Diese Installation enthält nur die Parser-Engine und ihre Abhängigkeiten, ohne Fetcher oder Kommandozeilenabhängigkeiten. Daher führt der Import von allem aus `scrapling.fetchers` oder `scrapling.spiders`, wie in den Beispielen oben, mit dieser Installation allein zu einem `ModuleNotFoundError`. Wenn Sie einen der Fetcher oder Spider verwenden möchten, installieren Sie zuerst die Fetcher-Abhängigkeiten wie unten gezeigt.
|
||||
|
||||
### Optionale Abhängigkeiten
|
||||
|
||||
|
||||
+16
-15
@@ -83,6 +83,15 @@ MySpider().start()
|
||||
|
||||
# Patrocinadores Platino
|
||||
<table>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank" title="Clean Proxies with No Nonsense.">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxidize.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank"><b>Proxidize</b></a> proporciona proxies móviles y residenciales para scraping, automatización de navegadores, monitoreo de SEO, agentes de IA y recopilación de datos. <i>Usa el código <b>scrapling20</b> para obtener un 20% de descuento</i>.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
@@ -137,16 +146,6 @@ MySpider().start()
|
||||
<a href="https://tikhub.io/?utm_source=github.com/D4Vinci/Scrapling&utm_medium=marketing_social&utm_campaign=retargeting&utm_content=carousel_ad" target="_blank">TikHub.io</a> ofrece más de 900 APIs estables en más de 16 plataformas, incluyendo TikTok, X, YouTube e Instagram, con más de 40M de conjuntos de datos. <br /> También ofrece <a href="https://ai.tikhub.io/?ref=KarimShoair" target="_blank">modelos de IA con descuento</a> - Claude, GPT, GEMINI y más con hasta un 71% de descuento.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank" title="Scalable Web Data Access for AI Applications">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/nsocks.png">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank">Nsocks</a> ofrece proxies residenciales e ISP rápidos para desarrolladores y scrapers. Cobertura IP global, alto anonimato, rotación inteligente y rendimiento fiable para automatización y extracción de datos. Usa <a href="https://www.xcrawl.com/?keyword=2p67aivg" target="_blank">Xcrawl</a> para simplificar el crawling web a gran escala.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://petrosky.io/d4vinci" target="_blank" title="PetroSky delivers cutting-edge VPS hosting.">
|
||||
@@ -185,17 +184,17 @@ MySpider().start()
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> ofrece proxies residenciales desde solo $0,015/IP o $0,68/GB. Más de 20 millones de IPs en más de 90 países. Sesiones fijas o rotativas, gestionadas desde la aplicación de escritorio o móvil.
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> ofrece proxies residenciales desde solo $0,018/IP o $0,68/GB. Más de 20 millones de IPs en más de 90 países. Sesiones fijas o rotativas, gestionadas desde la aplicación de escritorio o móvil.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.png">
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.svg" width="240" height="100">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank">NodeMaven</a> - proveedor de proxies confiable con la mayor calidad de IP del mercado. Usa el código promocional SCRAPLING35 para obtener un 35% de descuento en proxies.
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank">NodeMaven</a> - proveedor de proxies confiable con la mayor calidad de IP del mercado. Usa el código promocional SCRAPLING35 para obtener un 35% de descuento en proxies.
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
@@ -211,6 +210,7 @@ MySpider().start()
|
||||
<a href="https://proxyempire.io/?ref=scrapling&utm_source=scrapling" target="_blank" title="Collect The Data Your Project Needs with the Best Residential Proxies"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/ProxyEmpire.png"></a>
|
||||
<a href="https://www.webshare.io/?referral_code=48r2m2cd5uz1" target="_blank" title="The Most Reliable Proxy with Unparalleled Performance"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/webshare.png"></a>
|
||||
<a href="https://proxiware.com/?ref=scrapling" target="_blank" title="Collect Any Data. At Any Scale."><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxiware.png"></a>
|
||||
<a href="https://talordata.com/?campaignid=ZOYRsA4bX9BwWyO5&utm_source=D4Vinci&utm_term=Scrapling" target="_blank" title="Google SERP API for LLM, AI Agents & Full-Stack SEO"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/TalorData.jpg"></a>
|
||||
|
||||
|
||||
<!-- /sponsors -->
|
||||
@@ -477,7 +477,8 @@ Scrapling requiere Python 3.10 o superior:
|
||||
pip install scrapling
|
||||
```
|
||||
|
||||
Esta instalación solo incluye el motor de análisis y sus dependencias, sin ningún fetcher ni dependencias de línea de comandos.
|
||||
> [!IMPORTANT]
|
||||
> Esta instalación solo incluye el motor de análisis y sus dependencias, sin ningún fetcher ni dependencias de línea de comandos. Por lo tanto, importar cualquier cosa desde `scrapling.fetchers` o `scrapling.spiders`, como en los ejemplos anteriores, lanzará un `ModuleNotFoundError` solo con esta instalación. Si va a usar alguno de los fetchers o spiders, instale primero las dependencias de los fetchers como se muestra a continuación.
|
||||
|
||||
### Dependencias Opcionales
|
||||
|
||||
|
||||
+16
-15
@@ -83,6 +83,15 @@ MySpider().start()
|
||||
|
||||
# Sponsors Platine
|
||||
<table>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank" title="Clean Proxies with No Nonsense.">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxidize.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank"><b>Proxidize</b></a> fournit des proxies mobiles et résidentiels pour le scraping, l'automatisation de navigateur, le suivi SEO, les agents IA et la collecte de données. <i>Utilisez le code <b>scrapling20</b> pour bénéficier de 20% de réduction</i>.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
@@ -137,16 +146,6 @@ MySpider().start()
|
||||
<a href="https://tikhub.io/?utm_source=github.com/D4Vinci/Scrapling&utm_medium=marketing_social&utm_campaign=retargeting&utm_content=carousel_ad" target="_blank">TikHub.io</a> propose plus de 900 APIs stables sur plus de 16 plateformes, dont TikTok, X, YouTube et Instagram, avec plus de 40M de jeux de données. <br /> Propose également des <a href="https://ai.tikhub.io/?ref=KarimShoair" target="_blank">modèles IA à prix réduit</a> - Claude, GPT, GEMINI et plus, jusqu'à 71% de réduction.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank" title="Scalable Web Data Access for AI Applications">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/nsocks.png">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank">Nsocks</a> fournit des proxies résidentiels et ISP rapides pour les développeurs et les scrapeurs. Couverture IP mondiale, anonymat élevé, rotation intelligente et performances fiables pour l'automatisation et l'extraction de données. Utilisez <a href="https://www.xcrawl.com/?keyword=2p67aivg" target="_blank">Xcrawl</a> pour simplifier le crawling web à grande échelle.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://petrosky.io/d4vinci" target="_blank" title="PetroSky delivers cutting-edge VPS hosting.">
|
||||
@@ -185,17 +184,17 @@ MySpider().start()
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> propose des proxys résidentiels à partir de seulement 0,015 $/IP ou 0,68 $/Go. Plus de 20 millions d'IPs dans plus de 90 pays. Sessions fixes ou rotatives, gérées depuis l'application de bureau ou mobile.
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> propose des proxys résidentiels à partir de seulement 0,018 $/IP ou 0,68 $/Go. Plus de 20 millions d'IPs dans plus de 90 pays. Sessions fixes ou rotatives, gérées depuis l'application de bureau ou mobile.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.png">
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.svg" width="240" height="100">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank">NodeMaven</a> - fournisseur de proxys fiable offrant la meilleure qualité d'IP du marché. Utilisez le code promo SCRAPLING35 pour obtenir 35% de réduction sur les proxys.
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank">NodeMaven</a> - fournisseur de proxys fiable offrant la meilleure qualité d'IP du marché. Utilisez le code promo SCRAPLING35 pour obtenir 35% de réduction sur les proxys.
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
@@ -211,6 +210,7 @@ MySpider().start()
|
||||
<a href="https://proxyempire.io/?ref=scrapling&utm_source=scrapling" target="_blank" title="Collect The Data Your Project Needs with the Best Residential Proxies"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/ProxyEmpire.png"></a>
|
||||
<a href="https://www.webshare.io/?referral_code=48r2m2cd5uz1" target="_blank" title="The Most Reliable Proxy with Unparalleled Performance"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/webshare.png"></a>
|
||||
<a href="https://proxiware.com/?ref=scrapling" target="_blank" title="Collect Any Data. At Any Scale."><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxiware.png"></a>
|
||||
<a href="https://talordata.com/?campaignid=ZOYRsA4bX9BwWyO5&utm_source=D4Vinci&utm_term=Scrapling" target="_blank" title="Google SERP API for LLM, AI Agents & Full-Stack SEO"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/TalorData.jpg"></a>
|
||||
|
||||
|
||||
<!-- /sponsors -->
|
||||
@@ -477,7 +477,8 @@ Scrapling nécessite Python 3.10 ou supérieur :
|
||||
pip install scrapling
|
||||
```
|
||||
|
||||
Cette installation n'inclut que le moteur de parsing et ses dépendances, sans aucun fetcher ni dépendance en ligne de commande.
|
||||
> [!IMPORTANT]
|
||||
> Cette installation n'inclut que le moteur de parsing et ses dépendances, sans aucun fetcher ni dépendance en ligne de commande. Importer quoi que ce soit depuis `scrapling.fetchers` ou `scrapling.spiders`, comme dans les exemples ci-dessus, lèvera donc une `ModuleNotFoundError` avec cette seule installation. Si vous comptez utiliser l'un des fetchers ou spiders, installez d'abord les dépendances des fetchers comme indiqué ci-dessous.
|
||||
|
||||
### Dépendances optionnelles
|
||||
|
||||
|
||||
+16
-15
@@ -83,6 +83,15 @@ MySpider().start()
|
||||
|
||||
# プラチナスポンサー
|
||||
<table>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank" title="Clean Proxies with No Nonsense.">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxidize.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank"><b>Proxidize</b></a> は、スクレイピング、ブラウザ自動化、SEO監視、AIエージェント、データ収集のためのモバイルおよびレジデンシャルプロキシを提供します。<i>コード <b>scrapling20</b> で20%オフ</i>。
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
@@ -137,16 +146,6 @@ MySpider().start()
|
||||
<a href="https://tikhub.io/?utm_source=github.com/D4Vinci/Scrapling&utm_medium=marketing_social&utm_campaign=retargeting&utm_content=carousel_ad" target="_blank">TikHub.io</a> は TikTok、X、YouTube、Instagram を含む 16 以上のプラットフォームで 900 以上の安定した API を提供し、4,000 万以上のデータセットを保有。<br /> さらに <a href="https://ai.tikhub.io/?ref=KarimShoair" target="_blank">割引 AI モデル</a>も提供 - Claude、GPT、GEMINI など最大 71% オフ。
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank" title="Scalable Web Data Access for AI Applications">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/nsocks.png">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank">Nsocks</a> は開発者やスクレイパー向けの高速なレジデンシャルおよび ISP プロキシを提供。グローバル IP カバレッジ、高い匿名性、スマートなローテーション、自動化とデータ抽出のための信頼性の高いパフォーマンス。<a href="https://www.xcrawl.com/?keyword=2p67aivg" target="_blank">Xcrawl</a> で大規模ウェブクローリングを簡素化。
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://petrosky.io/d4vinci" target="_blank" title="PetroSky delivers cutting-edge VPS hosting.">
|
||||
@@ -185,17 +184,17 @@ MySpider().start()
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> はIPあたり $0.015 またはGBあたり $0.68 からの住宅用プロキシを提供します。90カ国以上で2,000万以上のIPを保有。固定セッションまたはローテーションセッションをデスクトップまたはモバイルアプリで管理できます。
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> はIPあたり $0.018 またはGBあたり $0.68 からの住宅用プロキシを提供します。90カ国以上で2,000万以上のIPを保有。固定セッションまたはローテーションセッションをデスクトップまたはモバイルアプリで管理できます。
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.png">
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.svg" width="240" height="100">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank">NodeMaven</a> - 市場最高品質のIPを提供する信頼性の高いプロキシプロバイダー。プロモコード SCRAPLING35 でプロキシが35%割引になります。
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank">NodeMaven</a> - 市場最高品質のIPを提供する信頼性の高いプロキシプロバイダー。プロモコード SCRAPLING35 でプロキシが35%割引になります。
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
@@ -211,6 +210,7 @@ MySpider().start()
|
||||
<a href="https://proxyempire.io/?ref=scrapling&utm_source=scrapling" target="_blank" title="Collect The Data Your Project Needs with the Best Residential Proxies"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/ProxyEmpire.png"></a>
|
||||
<a href="https://www.webshare.io/?referral_code=48r2m2cd5uz1" target="_blank" title="The Most Reliable Proxy with Unparalleled Performance"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/webshare.png"></a>
|
||||
<a href="https://proxiware.com/?ref=scrapling" target="_blank" title="Collect Any Data. At Any Scale."><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxiware.png"></a>
|
||||
<a href="https://talordata.com/?campaignid=ZOYRsA4bX9BwWyO5&utm_source=D4Vinci&utm_term=Scrapling" target="_blank" title="Google SERP API for LLM, AI Agents & Full-Stack SEO"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/TalorData.jpg"></a>
|
||||
|
||||
|
||||
<!-- /sponsors -->
|
||||
@@ -477,7 +477,8 @@ Scrapling には Python 3.10 以上が必要です:
|
||||
pip install scrapling
|
||||
```
|
||||
|
||||
このインストールにはパーサーエンジンとその依存関係のみが含まれており、Fetcher やコマンドライン依存関係は含まれていません。
|
||||
> [!IMPORTANT]
|
||||
> このインストールにはパーサーエンジンとその依存関係のみが含まれており、Fetcher やコマンドライン依存関係は含まれていません。 そのため、このインストールのみでは、上記の例のように `scrapling.fetchers` や `scrapling.spiders` から何かをインポートすると `ModuleNotFoundError` が発生します。Fetcher や Spider を使用する場合は、以下のように、まず Fetcher の依存関係をインストールしてください。
|
||||
|
||||
### オプションの依存関係
|
||||
|
||||
|
||||
+16
-15
@@ -83,6 +83,15 @@ MySpider().start()
|
||||
|
||||
# 플래티넘 스폰서
|
||||
<table>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank" title="Clean Proxies with No Nonsense.">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxidize.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank"><b>Proxidize</b></a>는 스크래핑, 브라우저 자동화, SEO 모니터링, AI 에이전트, 데이터 수집을 위한 모바일 및 주거용 프록시를 제공합니다. <i>코드 <b>scrapling20</b>으로 20% 할인</i>.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
@@ -137,16 +146,6 @@ MySpider().start()
|
||||
<a href="https://tikhub.io/?utm_source=github.com/D4Vinci/Scrapling&utm_medium=marketing_social&utm_campaign=retargeting&utm_content=carousel_ad" target="_blank">TikHub.io</a>는 TikTok, X, YouTube, Instagram 등 16개 이상 플랫폼에서 900개 이상의 안정적인 API를 제공하며, 4,000만 이상의 데이터셋을 보유하고 있습니다. <br /> <a href="https://ai.tikhub.io/?ref=KarimShoair" target="_blank">할인된 AI 모델</a>도 제공 - Claude, GPT, GEMINI 등 최대 71% 할인.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank" title="Scalable Web Data Access for AI Applications">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/nsocks.png">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank">Nsocks</a>는 개발자와 스크레이퍼를 위한 빠른 레지덴셜 및 ISP 프록시를 제공합니다. 글로벌 IP 커버리지, 높은 익명성, 스마트 로테이션, 자동화와 데이터 추출을 위한 안정적인 성능. <a href="https://www.xcrawl.com/?keyword=2p67aivg" target="_blank">Xcrawl</a>로 대규모 웹 크롤링을 간소화하세요.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://petrosky.io/d4vinci" target="_blank" title="PetroSky delivers cutting-edge VPS hosting.">
|
||||
@@ -185,17 +184,17 @@ MySpider().start()
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a>는 IP당 $0.015 또는 GB당 $0.68의 저렴한 가격부터 시작하는 주거용 프록시를 제공합니다. 90개국 이상에서 2천만 개 이상의 IP를 보유하고 있으며, 고정 또는 회전 세션을 데스크톱이나 모바일 앱에서 관리할 수 있습니다.
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a>는 IP당 $0.018 또는 GB당 $0.68의 저렴한 가격부터 시작하는 주거용 프록시를 제공합니다. 90개국 이상에서 2천만 개 이상의 IP를 보유하고 있으며, 고정 또는 회전 세션을 데스크톱이나 모바일 앱에서 관리할 수 있습니다.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.png">
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.svg" width="240" height="100">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank">NodeMaven</a> - 시장에서 가장 높은 품질의 IP를 제공하는 신뢰할 수 있는 프록시 제공업체입니다. 프로모 코드 SCRAPLING35를 사용하면 프록시 35% 할인을 받을 수 있습니다.
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank">NodeMaven</a> - 시장에서 가장 높은 품질의 IP를 제공하는 신뢰할 수 있는 프록시 제공업체입니다. 프로모 코드 SCRAPLING35를 사용하면 프록시 35% 할인을 받을 수 있습니다.
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
@@ -211,6 +210,7 @@ MySpider().start()
|
||||
<a href="https://proxyempire.io/?ref=scrapling&utm_source=scrapling" target="_blank" title="Collect The Data Your Project Needs with the Best Residential Proxies"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/ProxyEmpire.png"></a>
|
||||
<a href="https://www.webshare.io/?referral_code=48r2m2cd5uz1" target="_blank" title="The Most Reliable Proxy with Unparalleled Performance"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/webshare.png"></a>
|
||||
<a href="https://proxiware.com/?ref=scrapling" target="_blank" title="Collect Any Data. At Any Scale."><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxiware.png"></a>
|
||||
<a href="https://talordata.com/?campaignid=ZOYRsA4bX9BwWyO5&utm_source=D4Vinci&utm_term=Scrapling" target="_blank" title="Google SERP API for LLM, AI Agents & Full-Stack SEO"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/TalorData.jpg"></a>
|
||||
|
||||
|
||||
<!-- /sponsors -->
|
||||
@@ -477,7 +477,8 @@ Scrapling은 Python 3.10 이상이 필요합니다:
|
||||
pip install scrapling
|
||||
```
|
||||
|
||||
이 설치에는 파서 엔진과 의존성만 포함되며, Fetcher나 커맨드라인 의존성은 포함되지 않습니다.
|
||||
> [!IMPORTANT]
|
||||
> 이 설치에는 파서 엔진과 의존성만 포함되며, Fetcher나 커맨드라인 의존성은 포함되지 않습니다. 따라서 이 설치만으로는 위 예제처럼 `scrapling.fetchers`나 `scrapling.spiders`에서 무언가를 임포트하면 `ModuleNotFoundError`가 발생합니다. Fetcher나 Spider를 사용하려면 아래와 같이 먼저 Fetcher 의존성을 설치하세요.
|
||||
|
||||
### 선택적 의존성
|
||||
|
||||
|
||||
+16
-15
@@ -85,6 +85,15 @@ MySpider().start()
|
||||
|
||||
# Patrocinadores Platina
|
||||
<table>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank" title="Clean Proxies with No Nonsense.">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxidize.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> A <a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank"><b>Proxidize</b></a> oferece proxies móveis e residenciais para scraping, automação de navegador, monitoramento de SEO, agentes de IA e coleta de dados. <i>Use o código <b>scrapling20</b> para 20% de desconto</i>.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
@@ -139,16 +148,6 @@ MySpider().start()
|
||||
<a href="https://tikhub.io/?utm_source=github.com/D4Vinci/Scrapling&utm_medium=marketing_social&utm_campaign=retargeting&utm_content=carousel_ad" target="_blank">TikHub.io</a> oferece mais de 900 APIs estáveis em mais de 16 plataformas, incluindo TikTok, X, YouTube e Instagram, com mais de 40M de datasets. <br /> Também oferece <a href="https://ai.tikhub.io/?ref=KarimShoair" target="_blank">modelos de IA com desconto</a> - Claude, GPT, GEMINI e mais com até 71% de desconto.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank" title="Scalable Web Data Access for AI Applications">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/nsocks.png">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank">Nsocks</a> fornece proxies residenciais e ISP rápidos para desenvolvedores e scrapers. Cobertura global de IPs, alto anonimato, rotação inteligente e desempenho confiável para automação e extração de dados. Use o <a href="https://www.xcrawl.com/?keyword=2p67aivg" target="_blank">Xcrawl</a> para simplificar o crawling web em larga escala.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://petrosky.io/d4vinci" target="_blank" title="PetroSky delivers cutting-edge VPS hosting.">
|
||||
@@ -187,17 +186,17 @@ MySpider().start()
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> oferece proxies residenciais a partir de apenas $0,015/IP ou $0,68/GB. Mais de 20M de IPs em mais de 90 países. Sessões fixas ou rotativas, gerenciadas pelo aplicativo desktop ou móvel.
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> oferece proxies residenciais a partir de apenas $0,018/IP ou $0,68/GB. Mais de 20M de IPs em mais de 90 países. Sessões fixas ou rotativas, gerenciadas pelo aplicativo desktop ou móvel.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.png">
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.svg" width="240" height="100">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank">NodeMaven</a> - provedor de proxies confiável com a mais alta qualidade de IP do mercado. Use o código promocional SCRAPLING35 para obter 35% de desconto em proxies.
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank">NodeMaven</a> - provedor de proxies confiável com a mais alta qualidade de IP do mercado. Use o código promocional SCRAPLING35 para obter 35% de desconto em proxies.
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
@@ -213,6 +212,7 @@ MySpider().start()
|
||||
<a href="https://proxyempire.io/?ref=scrapling&utm_source=scrapling" target="_blank" title="Collect The Data Your Project Needs with the Best Residential Proxies"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/ProxyEmpire.png"></a>
|
||||
<a href="https://www.webshare.io/?referral_code=48r2m2cd5uz1" target="_blank" title="The Most Reliable Proxy with Unparalleled Performance"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/webshare.png"></a>
|
||||
<a href="https://proxiware.com/?ref=scrapling" target="_blank" title="Collect Any Data. At Any Scale."><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxiware.png"></a>
|
||||
<a href="https://talordata.com/?campaignid=ZOYRsA4bX9BwWyO5&utm_source=D4Vinci&utm_term=Scrapling" target="_blank" title="Google SERP API for LLM, AI Agents & Full-Stack SEO"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/TalorData.jpg"></a>
|
||||
|
||||
|
||||
<!-- /sponsors -->
|
||||
@@ -479,7 +479,8 @@ O Scrapling requer Python 3.10 ou superior:
|
||||
pip install scrapling
|
||||
```
|
||||
|
||||
Esta instalação inclui apenas o motor de parsing e suas dependências, sem fetchers nem dependências de linha de comando.
|
||||
> [!IMPORTANT]
|
||||
> Esta instalação inclui apenas o motor de parsing e suas dependências, sem fetchers nem dependências de linha de comando. Portanto, importar qualquer coisa de `scrapling.fetchers` ou `scrapling.spiders`, como nos exemplos acima, lançará um `ModuleNotFoundError` apenas com esta instalação. Se você for usar algum dos fetchers ou spiders, instale primeiro as dependências dos fetchers como mostrado abaixo.
|
||||
|
||||
### Dependências Opcionais
|
||||
|
||||
|
||||
+16
-15
@@ -83,6 +83,15 @@ MySpider().start()
|
||||
|
||||
# Платиновые спонсоры
|
||||
<table>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank" title="Clean Proxies with No Nonsense.">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxidize.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank"><b>Proxidize</b></a> предоставляет мобильные и резидентные прокси для скрейпинга, автоматизации браузера, SEO-мониторинга, ИИ-агентов и сбора данных. <i>Используйте код <b>scrapling20</b> для скидки 20%</i>.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
@@ -140,16 +149,6 @@ MySpider().start()
|
||||
<a href="https://tikhub.io/?utm_source=github.com/D4Vinci/Scrapling&utm_medium=marketing_social&utm_campaign=retargeting&utm_content=carousel_ad" target="_blank">TikHub.io</a> предоставляет более 900 стабильных API на 16+ платформах, включая TikTok, X, YouTube и Instagram, с более чем 40 млн наборов данных. <br /> Также предлагает <a href="https://ai.tikhub.io/?ref=KarimShoair" target="_blank">AI-модели со скидкой</a> - Claude, GPT, GEMINI и другие со скидкой до 71%.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank" title="Scalable Web Data Access for AI Applications">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/nsocks.png">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank">Nsocks</a> предоставляет быстрые резидентные и ISP прокси для разработчиков и скраперов. Глобальное покрытие IP, высокая анонимность, умная ротация и надёжная производительность для автоматизации и извлечения данных. Используйте <a href="https://www.xcrawl.com/?keyword=2p67aivg" target="_blank">Xcrawl</a> для упрощения масштабного веб-краулинга.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://petrosky.io/d4vinci" target="_blank" title="PetroSky delivers cutting-edge VPS hosting.">
|
||||
@@ -188,17 +187,17 @@ MySpider().start()
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> предоставляет резидентные прокси всего от $0,015 за IP или $0,68 за ГБ. Более 20 млн IP в 90+ странах. Закреплённые или ротационные сессии, управление через настольное или мобильное приложение.
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank">9Proxy</a> предоставляет резидентные прокси всего от $0,018 за IP или $0,68 за ГБ. Более 20 млн IP в 90+ странах. Закреплённые или ротационные сессии, управление через настольное или мобильное приложение.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.png">
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.svg" width="240" height="100">
|
||||
</a>
|
||||
</td>
|
||||
<td>
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank">NodeMaven</a> - надёжный провайдер прокси с самым высоким качеством IP на рынке. Используйте промокод SCRAPLING35 для получения скидки 35% на прокси.
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank">NodeMaven</a> - надёжный провайдер прокси с самым высоким качеством IP на рынке. Используйте промокод SCRAPLING35 для получения скидки 35% на прокси.
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
@@ -214,6 +213,7 @@ MySpider().start()
|
||||
<a href="https://proxyempire.io/?ref=scrapling&utm_source=scrapling" target="_blank" title="Collect The Data Your Project Needs with the Best Residential Proxies"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/ProxyEmpire.png"></a>
|
||||
<a href="https://www.webshare.io/?referral_code=48r2m2cd5uz1" target="_blank" title="The Most Reliable Proxy with Unparalleled Performance"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/webshare.png"></a>
|
||||
<a href="https://proxiware.com/?ref=scrapling" target="_blank" title="Collect Any Data. At Any Scale."><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxiware.png"></a>
|
||||
<a href="https://talordata.com/?campaignid=ZOYRsA4bX9BwWyO5&utm_source=D4Vinci&utm_term=Scrapling" target="_blank" title="Google SERP API for LLM, AI Agents & Full-Stack SEO"><img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/TalorData.jpg"></a>
|
||||
|
||||
|
||||
<!-- /sponsors -->
|
||||
@@ -480,7 +480,8 @@ Scrapling требует Python 3.10 или выше:
|
||||
pip install scrapling
|
||||
```
|
||||
|
||||
Эта установка включает только движок парсера и его зависимости, без каких-либо Fetcher'ов или зависимостей командной строки.
|
||||
> [!IMPORTANT]
|
||||
> Эта установка включает только движок парсера и его зависимости, без каких-либо Fetcher'ов или зависимостей командной строки. Поэтому импорт чего-либо из `scrapling.fetchers` или `scrapling.spiders`, как в примерах выше, вызовет `ModuleNotFoundError` при такой установке. Если вы собираетесь использовать какие-либо Fetcher'ы или Spider'ы, сначала установите зависимости Fetcher'ов, как показано ниже.
|
||||
|
||||
### Опциональные зависимости
|
||||
|
||||
|
||||
@@ -50,7 +50,7 @@ The extract command is a set of simple terminal tools that:
|
||||
scrapling extract get "https://example.com" content.txt
|
||||
|
||||
# Or use the Docker image with something like this:
|
||||
docker run -v $(pwd)/output:/output scrapling extract get "https://blog.example.com" /output/article.md
|
||||
docker run -v $(pwd)/output:/output pyd4vinci/scrapling extract get "https://blog.example.com" /output/article.md
|
||||
```
|
||||
|
||||
- **Extract Specific Content**
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
# Support and Advertisement
|
||||
|
||||
I've been creating all of these projects in my spare time and have invested considerable resources & effort in providing them to the community for free. By becoming a sponsor, you'd be directly funding my coffee reserves, helping me fulfill my responsibilities, and enabling me to continuously update existing projects and potentially create new ones.
|
||||
|
||||
You can sponsor me directly through the [GitHub Sponsors program](https://github.com/sponsors/D4Vinci) or [Buy Me a Coffee](https://buymeacoffee.com/d4vinci).
|
||||
|
||||
+8
-6
@@ -56,6 +56,9 @@ MySpider().start()
|
||||
|
||||
<!-- sponsors -->
|
||||
<div style="text-align: center;">
|
||||
<a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank" title="Clean Proxies with No Nonsense.">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxidize.png" class="ad">
|
||||
</a>
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/coldproxy.png" class="ad">
|
||||
</a>
|
||||
@@ -71,9 +74,6 @@ MySpider().start()
|
||||
<a href="https://tikhub.io/?utm_source=github.com/D4Vinci/Scrapling&utm_medium=marketing_social&utm_campaign=retargeting&utm_content=carousel_ad" target="_blank" title="Unlock the Power of Social Media Data & AI">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/TikHub.jpg" class="ad">
|
||||
</a>
|
||||
<a href="https://www.nsocks.com/?keyword=2p67aivg" target="_blank" title="Scalable Web Data Access for AI Applications">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/nsocks.png" class="ad">
|
||||
</a>
|
||||
<a href="https://petrosky.io/d4vinci" target="_blank" title="PetroSky delivers cutting-edge VPS hosting.">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/petrosky.png" class="ad">
|
||||
</a>
|
||||
@@ -86,8 +86,8 @@ MySpider().start()
|
||||
<a href="https://9proxy.com/pricing?tab=traffic&utm_source=Github&utm_campaign=D4vinci" target="_blank" title="Top-Tier Residential Proxy Solution for the Highest Success Rate">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/9proxy.jpg" class="ad">
|
||||
</a>
|
||||
<a href="https://go.nodemaven.com/scrapling" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.png">
|
||||
<a href="https://go.nodemaven.com/scraplingjune" target="_blank" title="Proxies with the Highest IP Scores">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/NodeMaven.svg" class="ad">
|
||||
</a>
|
||||
<br />
|
||||
<br />
|
||||
@@ -183,7 +183,9 @@ Scrapling requires Python 3.10 or higher:
|
||||
pip install scrapling
|
||||
```
|
||||
|
||||
This installation only includes the parser engine and its dependencies, without any fetchers or commandline dependencies.
|
||||
!!! warning
|
||||
|
||||
This installation only includes the parser engine and its dependencies, without any fetchers or commandline dependencies. So importing anything from `scrapling.fetchers` or `scrapling.spiders`, like in the examples above, will raise `ModuleNotFoundError` with this installation alone. If you are going to use any of the fetchers or spiders, install the fetchers' dependencies first as shown below.
|
||||
|
||||
### Optional Dependencies
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
zensical>=0.0.41
|
||||
zensical>=0.0.44
|
||||
mkdocstrings>=1.0.4
|
||||
mkdocstrings-python>=2.0.3
|
||||
mkdocstrings-python>=2.0.4
|
||||
griffe-inherited-docstrings>=1.1.3
|
||||
griffe-runtime-objects>=0.3.1
|
||||
griffe-sphinx>=0.2.1
|
||||
|
||||
Binary file not shown.
|
Before Width: | Height: | Size: 27 KiB |
File diff suppressed because one or more lines are too long
|
After Width: | Height: | Size: 29 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 5.2 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 22 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 2.3 KiB |
+4
-4
@@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta"
|
||||
[project]
|
||||
name = "scrapling"
|
||||
# Static version instead of a dynamic version so we can get better layer caching while building docker, check the docker file to understand
|
||||
version = "0.4.8"
|
||||
version = "0.4.9"
|
||||
description = "Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!"
|
||||
readme = {file = "README.md", content-type = "text/markdown"}
|
||||
license = {file = "LICENSE"}
|
||||
@@ -61,7 +61,7 @@ classifiers = [
|
||||
"Typing :: Typed",
|
||||
]
|
||||
dependencies = [
|
||||
"lxml>=6.1.0",
|
||||
"lxml>=6.1.1",
|
||||
"cssselect>=1.4.0",
|
||||
"orjson>=3.11.8",
|
||||
"tld>=0.13.2",
|
||||
@@ -73,8 +73,8 @@ dependencies = [
|
||||
fetchers = [
|
||||
"click>=8.3.0",
|
||||
"curl_cffi>=0.15.0",
|
||||
"playwright==1.59.0",
|
||||
"patchright==1.59.1",
|
||||
"playwright==1.60.0",
|
||||
"patchright==1.60.1",
|
||||
"browserforge>=1.2.4",
|
||||
"apify-fingerprint-datapoints>=0.13.0",
|
||||
"msgspec>=0.21.1",
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
__author__ = "Karim Shoair (karim.shoair@pm.me)"
|
||||
__version__ = "0.4.8"
|
||||
__version__ = "0.4.9"
|
||||
__copyright__ = "Copyright (c) 2024 Karim Shoair"
|
||||
|
||||
from typing import Any, TYPE_CHECKING
|
||||
|
||||
+3
-1
@@ -2,6 +2,7 @@ from pathlib import Path
|
||||
from subprocess import check_output
|
||||
from sys import executable as python_executable
|
||||
|
||||
from scrapling import __version__
|
||||
from scrapling.core.utils import log
|
||||
from scrapling.engines.toolbelt.custom import Response
|
||||
from scrapling.core.utils._shell import _CookieParser, _ParseHeaders
|
||||
@@ -10,7 +11,7 @@ from scrapling.core._types import List, Optional, Dict, Tuple, Any, Callable
|
||||
from orjson import loads as json_loads, JSONDecodeError
|
||||
|
||||
try:
|
||||
from click import command, option, Choice, group, argument
|
||||
from click import command, option, Choice, group, argument, version_option
|
||||
except (ImportError, ModuleNotFoundError) as e:
|
||||
raise ModuleNotFoundError(
|
||||
"You need to install scrapling with any of the extras to enable Shell commands. See: https://scrapling.readthedocs.io/en/latest/#installation"
|
||||
@@ -650,6 +651,7 @@ def stealthy_fetch(
|
||||
|
||||
|
||||
@group()
|
||||
@version_option(version=__version__, prog_name="Scrapling")
|
||||
def main():
|
||||
pass
|
||||
|
||||
|
||||
@@ -19,6 +19,7 @@ from scrapling.core._types import (
|
||||
Unpack,
|
||||
Optional,
|
||||
Awaitable,
|
||||
ProxyType,
|
||||
SUPPORTED_HTTP_METHODS,
|
||||
FollowRedirects,
|
||||
)
|
||||
@@ -244,10 +245,11 @@ class _SyncSessionLogic(_ConfigurationLogic):
|
||||
|
||||
try:
|
||||
for attempt in range(max_retries):
|
||||
proxy: Optional[ProxyType]
|
||||
if self._proxy_rotator and static_proxy is None:
|
||||
proxy = self._proxy_rotator.get_proxy()
|
||||
else:
|
||||
proxy = static_proxy
|
||||
proxy = static_proxy or self._default_proxy
|
||||
|
||||
request_args = self._merge_request_args(stealth=stealth, proxy=proxy, **kwargs)
|
||||
try:
|
||||
@@ -461,10 +463,11 @@ class _ASyncSessionLogic(_ConfigurationLogic):
|
||||
try:
|
||||
# Determine if we should use proxy rotation
|
||||
for attempt in range(max_retries):
|
||||
proxy: Optional[ProxyType]
|
||||
if self._proxy_rotator and static_proxy is None:
|
||||
proxy = self._proxy_rotator.get_proxy()
|
||||
else:
|
||||
proxy = static_proxy
|
||||
proxy = static_proxy or self._default_proxy
|
||||
|
||||
request_args = self._merge_request_args(stealth=stealth, proxy=proxy, **kwargs)
|
||||
try:
|
||||
|
||||
@@ -13,8 +13,8 @@ from scrapling.core._types import Dict, Literal, Tuple
|
||||
__OS_NAME__ = platform_system()
|
||||
OSName = Literal["linux", "macos", "windows"]
|
||||
# Current versions hardcoded for now (Playwright doesn't allow to know the version of a browser without launching it)
|
||||
chromium_version = 147
|
||||
chrome_version = 147
|
||||
chromium_version = 148
|
||||
chrome_version = 148
|
||||
|
||||
|
||||
@lru_cache(1, typed=True)
|
||||
|
||||
+4
-2
@@ -671,7 +671,7 @@ class Selector(SelectorsGeneration):
|
||||
element_data = self.retrieve(identifier or selector)
|
||||
if element_data:
|
||||
elements = self.relocate(element_data, percentage)
|
||||
if elements is not None and auto_save:
|
||||
if elements and auto_save:
|
||||
self.save(elements[0], identifier or selector)
|
||||
|
||||
return self.__handle_elements(elements)
|
||||
@@ -991,7 +991,9 @@ class Selector(SelectorsGeneration):
|
||||
SequenceMatcher(None, v, candidate_attributes.get(k, "")).ratio()
|
||||
for k, v in original_attributes.items()
|
||||
)
|
||||
checks += len(candidate_attributes)
|
||||
# Using `max` so candidates with extra attributes are penalized and candidates
|
||||
# with fewer attributes don't get inflated scores from a smaller denominator
|
||||
checks += max(len(original_attributes), len(candidate_attributes))
|
||||
else:
|
||||
if not candidate_attributes:
|
||||
# Both don't have attributes, this must mean something
|
||||
|
||||
@@ -64,7 +64,7 @@ class ResponseCacheManager:
|
||||
async with await anyio.open_file(temp_path, "wb") as f:
|
||||
await f.write(serialized)
|
||||
|
||||
await temp_path.rename(self._cache_path(fingerprint))
|
||||
await temp_path.replace(self._cache_path(fingerprint))
|
||||
except Exception as e:
|
||||
if await temp_path.exists():
|
||||
await temp_path.unlink()
|
||||
|
||||
@@ -50,7 +50,7 @@ class CheckpointManager:
|
||||
async with await anyio.open_file(temp_path, "wb") as f:
|
||||
await f.write(serialized)
|
||||
|
||||
await temp_path.rename(self._checkpoint_path)
|
||||
await temp_path.replace(self._checkpoint_path)
|
||||
|
||||
log.info(f"Checkpoint saved: {len(data.requests)} requests, {len(data.seen)} seen URLs")
|
||||
except Exception as e:
|
||||
|
||||
+2
-2
@@ -14,12 +14,12 @@
|
||||
"mimeType": "image/png"
|
||||
}
|
||||
],
|
||||
"version": "0.4.8",
|
||||
"version": "0.4.9",
|
||||
"packages": [
|
||||
{
|
||||
"registryType": "pypi",
|
||||
"identifier": "scrapling",
|
||||
"version": "0.4.8",
|
||||
"version": "0.4.9",
|
||||
"runtimeHint": "uvx",
|
||||
"packageArguments": [
|
||||
{
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[metadata]
|
||||
name = scrapling
|
||||
version = 0.4.8
|
||||
version = 0.4.9
|
||||
author = Karim Shoair
|
||||
author_email = karim.shoair@pm.me
|
||||
description = Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!
|
||||
|
||||
@@ -4,8 +4,9 @@ from unittest.mock import patch, MagicMock
|
||||
import pytest_httpbin
|
||||
|
||||
from scrapling.parser import Selector
|
||||
from scrapling import __version__
|
||||
from scrapling.cli import (
|
||||
shell, mcp, get, post, put, delete, fetch, stealthy_fetch
|
||||
main, shell, mcp, get, post, put, delete, fetch, stealthy_fetch
|
||||
)
|
||||
|
||||
|
||||
@@ -32,6 +33,12 @@ class TestCLI:
|
||||
def runner(self):
|
||||
return CliRunner()
|
||||
|
||||
def test_version_flag(self, runner):
|
||||
"""Test that the --version flag prints the Scrapling version and exits"""
|
||||
result = runner.invoke(main, ['--version'])
|
||||
assert result.exit_code == 0
|
||||
assert result.output.strip() == f'Scrapling, version {__version__}'
|
||||
|
||||
def test_shell_command(self, runner):
|
||||
"""Test shell command"""
|
||||
with patch('scrapling.core.shell.CustomShell') as mock_shell:
|
||||
|
||||
@@ -1,6 +1,9 @@
|
||||
import pytest
|
||||
from unittest.mock import patch, MagicMock, AsyncMock
|
||||
from curl_cffi.curl import CurlError
|
||||
|
||||
|
||||
from scrapling.engines.static import AsyncFetcherClient
|
||||
from scrapling.engines.static import _ASyncSessionLogic as AsyncFetcherSession, AsyncFetcherClient
|
||||
from scrapling.engines.toolbelt import ProxyRotator
|
||||
|
||||
|
||||
class TestFetcherSession:
|
||||
@@ -13,3 +16,47 @@ class TestFetcherSession:
|
||||
# Should not have context manager methods
|
||||
assert client.__aenter__ is None
|
||||
assert client.__aexit__ is None
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_session_level_proxy_is_applied(self):
|
||||
"""Session-level proxy must reach the request, not be silently dropped (#295)"""
|
||||
proxy = "http://10.255.255.1:9999"
|
||||
|
||||
async with AsyncFetcherSession(proxy=proxy) as session:
|
||||
with (
|
||||
patch.object(session._async_curl_session, "request", new=AsyncMock()) as mocked_request,
|
||||
patch("scrapling.engines.static.ResponseFactory.from_http_request", return_value=MagicMock()),
|
||||
):
|
||||
await session.get("http://example.com")
|
||||
|
||||
assert mocked_request.call_args.kwargs["proxy"] == proxy
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_per_request_proxy_overrides_session_proxy(self):
|
||||
"""A per-request proxy must take precedence over the session-level proxy"""
|
||||
request_proxy = "http://10.255.255.2:9999"
|
||||
|
||||
async with AsyncFetcherSession(proxy="http://10.255.255.1:9999") as session:
|
||||
with (
|
||||
patch.object(session._async_curl_session, "request", new=AsyncMock()) as mocked_request,
|
||||
patch("scrapling.engines.static.ResponseFactory.from_http_request", return_value=MagicMock()),
|
||||
):
|
||||
await session.get("http://example.com", proxy=request_proxy)
|
||||
|
||||
assert mocked_request.call_args.kwargs["proxy"] == request_proxy
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_proxy_rotates_per_retry_attempt(self):
|
||||
"""With a rotator, every retry attempt must pull a fresh proxy"""
|
||||
rotator = ProxyRotator(["http://p1:8080", "http://p2:8080"])
|
||||
|
||||
async with AsyncFetcherSession(proxy_rotator=rotator, retries=2, retry_delay=0) as session:
|
||||
with (
|
||||
patch.object(session._async_curl_session, "request", new=AsyncMock()) as mocked_request,
|
||||
patch("scrapling.engines.static.ResponseFactory.from_http_request", return_value=MagicMock()),
|
||||
):
|
||||
mocked_request.side_effect = [CurlError("transient"), MagicMock()]
|
||||
await session.get("http://example.com")
|
||||
|
||||
proxies_used = [call.kwargs["proxy"] for call in mocked_request.call_args_list]
|
||||
assert proxies_used == ["http://p1:8080", "http://p2:8080"]
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
import pytest
|
||||
|
||||
from unittest.mock import patch, MagicMock
|
||||
from curl_cffi.curl import CurlError
|
||||
|
||||
from scrapling.engines.static import _SyncSessionLogic as FetcherSession, FetcherClient
|
||||
from scrapling.engines.toolbelt import ProxyRotator
|
||||
|
||||
|
||||
class TestFetcherSession:
|
||||
@@ -9,11 +11,7 @@ class TestFetcherSession:
|
||||
|
||||
def test_fetcher_session_creation(self):
|
||||
"""Test FetcherSession creation"""
|
||||
session = FetcherSession(
|
||||
timeout=30,
|
||||
retries=3,
|
||||
stealthy_headers=True
|
||||
)
|
||||
session = FetcherSession(timeout=30, retries=3, stealthy_headers=True)
|
||||
|
||||
assert session._default_timeout == 30
|
||||
assert session._default_retries == 3
|
||||
@@ -43,3 +41,44 @@ class TestFetcherSession:
|
||||
# Should not have context manager methods
|
||||
assert client.__enter__ is None
|
||||
assert client.__exit__ is None
|
||||
|
||||
def test_session_level_proxy_is_applied(self):
|
||||
"""Session-level proxy must reach the request, not be silently dropped (#295)"""
|
||||
proxy = "http://10.255.255.1:9999"
|
||||
|
||||
with FetcherSession(proxy=proxy) as session:
|
||||
with (
|
||||
patch.object(session._curl_session, "request") as mocked_request,
|
||||
patch("scrapling.engines.static.ResponseFactory.from_http_request", return_value=MagicMock()),
|
||||
):
|
||||
session.get("http://example.com")
|
||||
|
||||
assert mocked_request.call_args.kwargs["proxy"] == proxy
|
||||
|
||||
def test_per_request_proxy_overrides_session_proxy(self):
|
||||
"""A per-request proxy must take precedence over the session-level proxy"""
|
||||
request_proxy = "http://10.255.255.2:9999"
|
||||
|
||||
with FetcherSession(proxy="http://10.255.255.1:9999") as session:
|
||||
with (
|
||||
patch.object(session._curl_session, "request") as mocked_request,
|
||||
patch("scrapling.engines.static.ResponseFactory.from_http_request", return_value=MagicMock()),
|
||||
):
|
||||
session.get("http://example.com", proxy=request_proxy)
|
||||
|
||||
assert mocked_request.call_args.kwargs["proxy"] == request_proxy
|
||||
|
||||
def test_proxy_rotates_per_retry_attempt(self):
|
||||
"""With a rotator, every retry attempt must pull a fresh proxy"""
|
||||
rotator = ProxyRotator(["http://p1:8080", "http://p2:8080"])
|
||||
|
||||
with FetcherSession(proxy_rotator=rotator, retries=2, retry_delay=0) as session:
|
||||
with (
|
||||
patch.object(session._curl_session, "request") as mocked_request,
|
||||
patch("scrapling.engines.static.ResponseFactory.from_http_request", return_value=MagicMock()),
|
||||
):
|
||||
mocked_request.side_effect = [CurlError("transient"), MagicMock()]
|
||||
session.get("http://example.com")
|
||||
|
||||
proxies_used = [call.kwargs["proxy"] for call in mocked_request.call_args_list]
|
||||
assert proxies_used == ["http://p1:8080", "http://p2:8080"]
|
||||
|
||||
@@ -56,6 +56,32 @@ class TestParserAdaptive:
|
||||
assert relocated[0].has_class("new-class")
|
||||
assert relocated[0].css(".new-description")[0].text == "Description 1"
|
||||
|
||||
def test_relocation_auto_save_no_match_above_threshold(self):
|
||||
"""Adaptive relocation with `auto_save=True` must not crash when no element
|
||||
clears the `percentage` threshold (relocate() returns an empty list)."""
|
||||
original_html = """
|
||||
<div class="container">
|
||||
<article class="product" id="target">
|
||||
<h3>Widget</h3>
|
||||
<p class="desc">A widget</p>
|
||||
</article>
|
||||
</div>
|
||||
"""
|
||||
# Unrelated structure so nothing can match a high threshold
|
||||
changed_html = "<html><body><span>totally unrelated content</span></body></html>"
|
||||
|
||||
old_page = Selector(original_html, url="example.com", adaptive=True)
|
||||
new_page = Selector(changed_html, url="example.com", adaptive=True)
|
||||
|
||||
old_page.css("#target", identifier="target", auto_save=True)
|
||||
|
||||
# Before the fix this raised `IndexError: list index out of range` because the
|
||||
# guard checked `elements is not None` but relocate() returns [] (never None).
|
||||
result = new_page.css(
|
||||
"#target", identifier="target", adaptive=True, auto_save=True, percentage=95
|
||||
)
|
||||
assert list(result) == []
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_element_relocation_async(self):
|
||||
"""Test relocating element after structure change in async mode"""
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
Tests for Selector.find_similar() with non-default parameters.
|
||||
Target file: tests/parser/test_general.py (append to TestSimilarElements class)
|
||||
"""
|
||||
|
||||
import pytest
|
||||
from scrapling import Selector
|
||||
|
||||
@@ -61,14 +62,10 @@ class TestFindSimilarAdvanced:
|
||||
first = product_page.css("div.product")[0]
|
||||
# Ignore both data-price and data-category → only class matters → all 3 divs match
|
||||
ignore_all_data = first.find_similar(
|
||||
similarity_threshold=0.2,
|
||||
ignore_attributes=["data-price", "data-category"]
|
||||
similarity_threshold=0.2, ignore_attributes=["data-price", "data-category"]
|
||||
)
|
||||
# Ignore nothing → data-category difference (fruit vs veggie) may reduce matches
|
||||
ignore_nothing = first.find_similar(
|
||||
similarity_threshold=0.9,
|
||||
ignore_attributes=[]
|
||||
)
|
||||
ignore_nothing = first.find_similar(similarity_threshold=0.9, ignore_attributes=[])
|
||||
assert len(ignore_all_data) >= len(ignore_nothing)
|
||||
|
||||
def test_find_similar_on_text_node_returns_empty(self, product_page):
|
||||
@@ -76,3 +73,31 @@ class TestFindSimilarAdvanced:
|
||||
text_node = product_page.css(".name::text")[0]
|
||||
result = text_node.find_similar()
|
||||
assert len(result) == 0
|
||||
|
||||
def test_find_similar_attribute_count_mismatch_scoring(self):
|
||||
"""The similarity denominator uses max() of both attribute counts, so candidates
|
||||
with fewer attributes don't get inflated scores and candidates with extra
|
||||
attributes stay penalized."""
|
||||
html = """
|
||||
<html><body>
|
||||
<div class="cards">
|
||||
<div class="card" data-kind="primary" data-color="red" data-size="large">Alpha</div>
|
||||
<div class="card">Beta</div>
|
||||
<div class="card" data-kind="primary" data-color="red" data-size="large" data-id="x">Gamma</div>
|
||||
<div class="card" data-kind="primary" data-color="red" data-size="large">Delta</div>
|
||||
</div>
|
||||
</body></html>
|
||||
"""
|
||||
page = Selector(html, adaptive=False)
|
||||
first = page.css("div.card")[0] # Alpha
|
||||
|
||||
similar = first.find_similar(similarity_threshold=0.9, ignore_attributes=[])
|
||||
texts = {el.text for el in similar}
|
||||
|
||||
# An exact attribute match must pass
|
||||
assert "Delta" in texts
|
||||
# Beta matches 1 of Alpha's 4 attributes; the old denominator counted candidate
|
||||
# attributes only, inflating it to a perfect score (1.0 / 1)
|
||||
assert "Beta" not in texts
|
||||
# Gamma's extra attribute dilutes the score (4.0 / 5) - the intentional penalty
|
||||
assert "Gamma" not in texts
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
pytest>=2.8.0,<9
|
||||
pytest-cov
|
||||
playwright==1.59.0
|
||||
playwright==1.60.0
|
||||
werkzeug<3.0.0
|
||||
pytest-httpbin==2.1.0
|
||||
pytest-asyncio
|
||||
|
||||
@@ -0,0 +1,196 @@
|
||||
"""Boss直聘岗位爬虫 - headful模式,支持手动登录后爬取
|
||||
|
||||
使用方法:
|
||||
1. 第一次运行时,浏览器会弹出Boss直聘登录页,手动扫码/登录
|
||||
2. 登录成功后按回车键继续爬取
|
||||
3. 登录状态会保存到 user_data_dir,下次运行可跳过登录
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
|
||||
from scrapling.fetchers import StealthyFetcher
|
||||
from scrapling.core._types import Any, Dict
|
||||
|
||||
|
||||
# ===== 配置 =====
|
||||
SEARCH_KEYWORD = "Python"
|
||||
SEARCH_CITY = "100010000" # 全国
|
||||
USER_DATA_DIR = os.path.join(os.path.dirname(__file__), "boss_browser_data")
|
||||
OUTPUT_FILE = os.path.join(os.path.dirname(__file__), "boss_zhipin_jobs.json")
|
||||
MAX_PAGES = 5
|
||||
|
||||
|
||||
def crawl_boss_zhipin():
|
||||
"""使用StealthyFetcher爬取Boss直聘岗位信息"""
|
||||
|
||||
# 构建搜索URL
|
||||
url = f"https://www.zhipin.com/web/geek/job?query={SEARCH_KEYWORD}&city={SEARCH_CITY}"
|
||||
print(f"目标URL: {url}")
|
||||
print(f"关键词: {SEARCH_KEYWORD} | 城市: {SEARCH_CITY}")
|
||||
print(f"最大页数: {MAX_PAGES}")
|
||||
print("-" * 60)
|
||||
|
||||
all_jobs = []
|
||||
|
||||
for page_num in range(1, MAX_PAGES + 1):
|
||||
page_url = f"{url}&page={page_num}"
|
||||
print(f"\n正在爬取第 {page_num}/{MAX_PAGES} 页...")
|
||||
|
||||
try:
|
||||
response = StealthyFetcher.fetch(
|
||||
page_url,
|
||||
headless=False, # headful模式,方便登录
|
||||
network_idle=True,
|
||||
wait=3000,
|
||||
timeout=90000,
|
||||
block_ads=True,
|
||||
disable_resources=True,
|
||||
locale="zh-CN",
|
||||
timezone_id="Asia/Shanghai",
|
||||
user_data_dir=USER_DATA_DIR, # 保存浏览器会话
|
||||
real_chrome=True, # 使用本机Chrome
|
||||
)
|
||||
|
||||
print(f" 状态码: {response.status}")
|
||||
print(f" 最终URL: {response.url}")
|
||||
|
||||
# 检查是否被重定向到登录页
|
||||
if "/web/user/" in str(response.url) or "注册登录" in (response.css("title::text").get() or ""):
|
||||
print(f"\n⚠️ 页面被重定向到登录页!")
|
||||
print(f" 浏览器窗口已打开,请在浏览器中完成登录(扫码/账号登录)。")
|
||||
print(f" 登录成功后,请在此处按回车键继续...")
|
||||
input(" >>> 按回车继续 <<<")
|
||||
|
||||
# 重新抓取
|
||||
print(f"\n 重新爬取第 {page_num} 页...")
|
||||
response = StealthyFetcher.fetch(
|
||||
page_url,
|
||||
headless=False,
|
||||
network_idle=True,
|
||||
wait=5000,
|
||||
timeout=90000,
|
||||
block_ads=True,
|
||||
disable_resources=True,
|
||||
locale="zh-CN",
|
||||
timezone_id="Asia/Shanghai",
|
||||
user_data_dir=USER_DATA_DIR,
|
||||
real_chrome=True,
|
||||
)
|
||||
print(f" 状态码: {response.status}")
|
||||
print(f" 最终URL: {response.url}")
|
||||
|
||||
# 解析岗位卡片
|
||||
job_cards = response.css(".job-card-wrapper")
|
||||
if not job_cards:
|
||||
job_cards = response.css('[class*="job-card"]')
|
||||
if not job_cards:
|
||||
# 尝试更通用的选择器
|
||||
job_cards = response.css('[data-type="job"]')
|
||||
|
||||
print(f" 找到 {len(job_cards)} 个岗位")
|
||||
|
||||
if not job_cards:
|
||||
# 调试: 打印页面信息
|
||||
title = response.css("title::text").get() or ""
|
||||
print(f" 页面标题: {title}")
|
||||
# 保存当前页面HTML供调试
|
||||
debug_path = os.path.join(os.path.dirname(__file__), f"boss_page_{page_num}_debug.html")
|
||||
with open(debug_path, "w", encoding="utf-8") as f:
|
||||
f.write(response.body.decode("utf-8", errors="ignore"))
|
||||
print(f" 页面HTML已保存到: {debug_path}")
|
||||
|
||||
if page_num == 1:
|
||||
print("\n 第一页未找到岗位,可能需要登录。")
|
||||
print(" 请在浏览器中完成登录后按回车重试...")
|
||||
input(" >>> 按回车继续 <<<")
|
||||
continue
|
||||
else:
|
||||
break
|
||||
|
||||
page_jobs = 0
|
||||
for card in job_cards:
|
||||
try:
|
||||
job_name = card.css(".job-name::text").get() or card.css('[class*="job-name"]::text').get() or ""
|
||||
salary = card.css(".salary::text").get() or card.css('[class*="salary"]::text').get() or ""
|
||||
company_name = (
|
||||
card.css(".company-name::text").get()
|
||||
or card.css('[class*="company-name"] a::text').get()
|
||||
or card.css(".company-name a::text").get()
|
||||
or ""
|
||||
)
|
||||
area = card.css(".job-area::text").get() or card.css('[class*="job-area"]::text').get() or ""
|
||||
tag_list = (
|
||||
card.css(".tag-list li::text").getall()
|
||||
or card.css('[class*="tag-list"]::text').getall()
|
||||
or []
|
||||
)
|
||||
job_link = card.css(".job-card-left::attr(href)").get() or card.css("a::attr(href)").get() or ""
|
||||
|
||||
# 获取经验/学历要求
|
||||
info_items = card.css(".job-info .tag-list li::text").getall() or []
|
||||
if not info_items:
|
||||
info_items = card.css('[class*="info"] li::text').getall() or []
|
||||
|
||||
item = {
|
||||
"job_name": job_name.strip() if job_name else "",
|
||||
"salary": salary.strip() if salary else "",
|
||||
"company_name": company_name.strip() if company_name else "",
|
||||
"area": area.strip() if area else "",
|
||||
"tags": [t.strip() for t in tag_list if t.strip()],
|
||||
"requirements": [t.strip() for t in info_items if t.strip()],
|
||||
"job_url": response.urljoin(job_link) if job_link else "",
|
||||
"page": page_num,
|
||||
}
|
||||
|
||||
if item["job_name"]:
|
||||
all_jobs.append(item)
|
||||
page_jobs += 1
|
||||
print(f" [{page_jobs}] {item['job_name']} | {item['salary']} | {item['company_name']}")
|
||||
except Exception as e:
|
||||
print(f" 解析岗位出错: {e}")
|
||||
|
||||
print(f" 第 {page_num} 页成功解析 {page_jobs} 个岗位")
|
||||
|
||||
if page_jobs == 0 and page_num > 1:
|
||||
print(" 没有更多岗位,停止翻页")
|
||||
break
|
||||
|
||||
# 翻页延迟
|
||||
if page_num < MAX_PAGES:
|
||||
delay = 3
|
||||
print(f" 等待 {delay}s 后翻页...")
|
||||
time.sleep(delay)
|
||||
|
||||
except Exception as e:
|
||||
print(f" 爬取出错: {e}")
|
||||
import traceback
|
||||
traceback.print_exc()
|
||||
break
|
||||
|
||||
# 输出结果
|
||||
print("\n" + "=" * 60)
|
||||
print(f"爬取完成!")
|
||||
print(f" 总岗位数: {len(all_jobs)}")
|
||||
print("=" * 60)
|
||||
|
||||
for i, item in enumerate(all_jobs[:30]):
|
||||
print(f"\n[{i+1}] {item['job_name']}")
|
||||
print(f" 薪资: {item['salary']}")
|
||||
print(f" 公司: {item['company_name']}")
|
||||
print(f" 地区: {item['area']}")
|
||||
print(f" 标签: {', '.join(item.get('tags', []))}")
|
||||
print(f" 要求: {', '.join(item.get('requirements', []))}")
|
||||
print(f" 链接: {item['job_url']}")
|
||||
|
||||
# 保存到JSON
|
||||
with open(OUTPUT_FILE, "w", encoding="utf-8") as f:
|
||||
json.dump(all_jobs, f, ensure_ascii=False, indent=2)
|
||||
print(f"\n结果已保存到: {OUTPUT_FILE}")
|
||||
|
||||
return all_jobs
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
jobs = crawl_boss_zhipin()
|
||||
@@ -49,6 +49,27 @@ class TestResponseCacheManager:
|
||||
assert dict(restored.headers) == dict(original.headers)
|
||||
assert dict(restored.request_headers) == dict(original.request_headers)
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_put_overwrites_existing_entry(self):
|
||||
"""Re-caching the same fingerprint must replace the stored response.
|
||||
|
||||
Regression test for a Windows-only failure: ``Path.rename`` cannot
|
||||
overwrite an existing destination on Windows (raising ``WinError 183``),
|
||||
so the second ``put`` was caught by the error handler, the temp file was
|
||||
removed, and ``get`` kept returning the stale body. ``Path.replace``
|
||||
overwrites atomically on every platform.
|
||||
"""
|
||||
with tempfile.TemporaryDirectory() as tmpdir:
|
||||
cache = ResponseCacheManager(tmpdir)
|
||||
fp = b"\x05" * 20
|
||||
|
||||
await cache.put(fp, _make_response(body=b"<html>first</html>"), "GET")
|
||||
await cache.put(fp, _make_response(body=b"<html>second</html>"), "GET")
|
||||
|
||||
restored = await cache.get(fp)
|
||||
assert restored is not None
|
||||
assert restored.body == b"<html>second</html>"
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_get_cache_miss(self):
|
||||
with tempfile.TemporaryDirectory() as tmpdir:
|
||||
|
||||
Reference in New Issue
Block a user