mirror of
https://github.com/D4Vinci/Scrapling.git
synced 2026-09-14 20:07:02 +08:00
Merge branch 'dev' into fix/cache-drops-browser-cookies
This commit is contained in:
@@ -20,6 +20,8 @@
|
||||
<img alt="Tests" src="https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml/badge.svg"></a>
|
||||
<a href="https://badge.fury.io/py/Scrapling" alt="PyPI version">
|
||||
<img alt="PyPI version" src="https://badge.fury.io/py/Scrapling.svg"></a>
|
||||
<a href="https://hub.docker.com/r/pyd4vinci/scrapling" target="_blank">
|
||||
<img alt="Docker Pulls" src="https://img.shields.io/docker/pulls/pyd4vinci/scrapling?labelColor=%20%23FDB062&logo=Docker&labelColor=%20%23528bff"></a>
|
||||
<a href="https://clickpy.clickhouse.com/dashboard/scrapling" rel="nofollow"><img src="https://img.shields.io/pypi/dm/scrapling" alt="PyPI package downloads"></a>
|
||||
<a href="https://github.com/D4Vinci/Scrapling/tree/main/agent-skill" alt="AI Agent Skill directory">
|
||||
<img alt="Static Badge" src="https://img.shields.io/badge/Skill-black?style=flat&label=Agent&link=https%3A%2F%2Fgithub.com%2FD4Vinci%2FScrapling%2Ftree%2Fmain%2Fagent-skill"></a>
|
||||
@@ -98,11 +100,11 @@ MySpider().start()
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/coldproxy.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://coldproxy.com/" target="_blank"><b>ColdProxy</b></a> provides residential and datacenter proxies for stable web scraping, public data collection, and geo-targeted testing across 195+ countries.
|
||||
<td> <a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank"><b>ColdProxy</b></a> provides residential and datacenter proxies for stable web scraping, public data collection, and geo-targeted testing across 195+ countries.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
name: scrapling-official
|
||||
description: Scrape web pages using Scrapling with anti-bot bypass (like Cloudflare Turnstile), stealth headless browsing, spiders framework, adaptive scraping, and JavaScript rendering. Use when asked to scrape, crawl, or extract data from websites; web_fetch fails; the site has anti-bot protections; write Python code to scrape/crawl; or write spiders.
|
||||
version: "0.4.11"
|
||||
version: "0.4.12"
|
||||
license: Complete terms in LICENSE.txt
|
||||
metadata:
|
||||
homepage: "https://scrapling.readthedocs.io/en/latest/index.html"
|
||||
@@ -40,7 +40,7 @@ Blazing fast crawls with real-time stats and streaming. Built by Web Scrapers fo
|
||||
|
||||
Create a virtual Python environment through any way available, like `venv`, then inside the environment do:
|
||||
|
||||
`pip install "scrapling[all]>=0.4.11"`
|
||||
`pip install "scrapling[all]>=0.4.12"`
|
||||
|
||||
Then do this to download all the browsers' dependencies:
|
||||
|
||||
|
||||
@@ -9,7 +9,7 @@ All examples collect **all 100 quotes across 10 pages**.
|
||||
Make sure Scrapling is installed:
|
||||
|
||||
```bash
|
||||
pip install "scrapling[all]>=0.4.11"
|
||||
pip install "scrapling[all]>=0.4.12"
|
||||
scrapling install --force
|
||||
```
|
||||
|
||||
|
||||
+4
-2
@@ -16,6 +16,8 @@
|
||||
<img alt="Tests" src="https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml/badge.svg"></a>
|
||||
<a href="https://badge.fury.io/py/Scrapling" alt="PyPI version">
|
||||
<img alt="PyPI version" src="https://badge.fury.io/py/Scrapling.svg"></a>
|
||||
<a href="https://hub.docker.com/r/pyd4vinci/scrapling" target="_blank">
|
||||
<img alt="Docker Pulls" src="https://img.shields.io/docker/pulls/pyd4vinci/scrapling?labelColor=%20%23FDB062&logo=Docker&labelColor=%20%23528bff"></a>
|
||||
<a href="https://clickpy.clickhouse.com/dashboard/scrapling" rel="nofollow"><img src="https://img.shields.io/pypi/dm/scrapling" alt="PyPI package downloads"></a>
|
||||
<a href="https://github.com/D4Vinci/Scrapling/tree/main/agent-skill" alt="AI Agent Skill directory">
|
||||
<img alt="Static Badge" src="https://img.shields.io/badge/Skill-black?style=flat&label=Agent&link=https%3A%2F%2Fgithub.com%2FD4Vinci%2FScrapling%2Ftree%2Fmain%2Fagent-skill"></a>
|
||||
@@ -94,11 +96,11 @@ MySpider().start()
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/coldproxy.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> توفر <a href="https://coldproxy.com/" target="_blank"><b>ColdProxy</b></a> وكلاء سكنيين ووكلاء مراكز بيانات لاستخراج بيانات الويب بشكل مستقر، وجمع البيانات العامة، والاختبار الموجَّه جغرافياً في أكثر من 195 دولة.
|
||||
<td> توفر <a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank"><b>ColdProxy</b></a> وكلاء سكنيين ووكلاء مراكز بيانات لاستخراج بيانات الويب بشكل مستقر، وجمع البيانات العامة، والاختبار الموجَّه جغرافياً في أكثر من 195 دولة.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
|
||||
+4
-2
@@ -16,6 +16,8 @@
|
||||
<img alt="Tests" src="https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml/badge.svg"></a>
|
||||
<a href="https://badge.fury.io/py/Scrapling" alt="PyPI version">
|
||||
<img alt="PyPI version" src="https://badge.fury.io/py/Scrapling.svg"></a>
|
||||
<a href="https://hub.docker.com/r/pyd4vinci/scrapling" target="_blank">
|
||||
<img alt="Docker Pulls" src="https://img.shields.io/docker/pulls/pyd4vinci/scrapling?labelColor=%20%23FDB062&logo=Docker&labelColor=%20%23528bff"></a>
|
||||
<a href="https://clickpy.clickhouse.com/dashboard/scrapling" rel="nofollow"><img src="https://img.shields.io/pypi/dm/scrapling" alt="PyPI package downloads"></a>
|
||||
<a href="https://github.com/D4Vinci/Scrapling/tree/main/agent-skill" alt="AI Agent Skill directory">
|
||||
<img alt="Static Badge" src="https://img.shields.io/badge/Skill-black?style=flat&label=Agent&link=https%3A%2F%2Fgithub.com%2FD4Vinci%2FScrapling%2Ftree%2Fmain%2Fagent-skill"></a>
|
||||
@@ -94,11 +96,11 @@ MySpider().start()
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/coldproxy.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://coldproxy.com/" target="_blank"><b>ColdProxy</b></a> 提供住宅代理和数据中心代理,用于稳定的网络抓取、公共数据收集,以及覆盖 195 多个国家/地区的地理定向测试。
|
||||
<td> <a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank"><b>ColdProxy</b></a> 提供住宅代理和数据中心代理,用于稳定的网络抓取、公共数据收集,以及覆盖 195 多个国家/地区的地理定向测试。
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
|
||||
+4
-2
@@ -16,6 +16,8 @@
|
||||
<img alt="Tests" src="https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml/badge.svg"></a>
|
||||
<a href="https://badge.fury.io/py/Scrapling" alt="PyPI version">
|
||||
<img alt="PyPI version" src="https://badge.fury.io/py/Scrapling.svg"></a>
|
||||
<a href="https://hub.docker.com/r/pyd4vinci/scrapling" target="_blank">
|
||||
<img alt="Docker Pulls" src="https://img.shields.io/docker/pulls/pyd4vinci/scrapling?labelColor=%20%23FDB062&logo=Docker&labelColor=%20%23528bff"></a>
|
||||
<a href="https://clickpy.clickhouse.com/dashboard/scrapling" rel="nofollow"><img src="https://img.shields.io/pypi/dm/scrapling" alt="PyPI package downloads"></a>
|
||||
<a href="https://github.com/D4Vinci/Scrapling/tree/main/agent-skill" alt="AI Agent Skill directory">
|
||||
<img alt="Static Badge" src="https://img.shields.io/badge/Skill-black?style=flat&label=Agent&link=https%3A%2F%2Fgithub.com%2FD4Vinci%2FScrapling%2Ftree%2Fmain%2Fagent-skill"></a>
|
||||
@@ -94,11 +96,11 @@ MySpider().start()
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/coldproxy.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://coldproxy.com/" target="_blank"><b>ColdProxy</b></a> bietet Residential- und Datacenter-Proxies für stabiles Web Scraping, öffentliche Datenerfassung und geografisch gezielte Tests in über 195 Ländern.
|
||||
<td> <a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank"><b>ColdProxy</b></a> bietet Residential- und Datacenter-Proxies für stabiles Web Scraping, öffentliche Datenerfassung und geografisch gezielte Tests in über 195 Ländern.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
|
||||
+4
-2
@@ -16,6 +16,8 @@
|
||||
<img alt="Tests" src="https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml/badge.svg"></a>
|
||||
<a href="https://badge.fury.io/py/Scrapling" alt="PyPI version">
|
||||
<img alt="PyPI version" src="https://badge.fury.io/py/Scrapling.svg"></a>
|
||||
<a href="https://hub.docker.com/r/pyd4vinci/scrapling" target="_blank">
|
||||
<img alt="Docker Pulls" src="https://img.shields.io/docker/pulls/pyd4vinci/scrapling?labelColor=%20%23FDB062&logo=Docker&labelColor=%20%23528bff"></a>
|
||||
<a href="https://clickpy.clickhouse.com/dashboard/scrapling" rel="nofollow"><img src="https://img.shields.io/pypi/dm/scrapling" alt="PyPI package downloads"></a>
|
||||
<a href="https://github.com/D4Vinci/Scrapling/tree/main/agent-skill" alt="AI Agent Skill directory">
|
||||
<img alt="Static Badge" src="https://img.shields.io/badge/Skill-black?style=flat&label=Agent&link=https%3A%2F%2Fgithub.com%2FD4Vinci%2FScrapling%2Ftree%2Fmain%2Fagent-skill"></a>
|
||||
@@ -94,11 +96,11 @@ MySpider().start()
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/coldproxy.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://coldproxy.com/" target="_blank"><b>ColdProxy</b></a> proporciona proxies residenciales y de centros de datos para web scraping estable, recopilación de datos públicos y pruebas con segmentación geográfica en más de 195 países.
|
||||
<td> <a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank"><b>ColdProxy</b></a> proporciona proxies residenciales y de centros de datos para web scraping estable, recopilación de datos públicos y pruebas con segmentación geográfica en más de 195 países.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
|
||||
+4
-2
@@ -16,6 +16,8 @@
|
||||
<img alt="Tests" src="https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml/badge.svg"></a>
|
||||
<a href="https://badge.fury.io/py/Scrapling" alt="PyPI version">
|
||||
<img alt="PyPI version" src="https://badge.fury.io/py/Scrapling.svg"></a>
|
||||
<a href="https://hub.docker.com/r/pyd4vinci/scrapling" target="_blank">
|
||||
<img alt="Docker Pulls" src="https://img.shields.io/docker/pulls/pyd4vinci/scrapling?labelColor=%20%23FDB062&logo=Docker&labelColor=%20%23528bff"></a>
|
||||
<a href="https://clickpy.clickhouse.com/dashboard/scrapling" rel="nofollow"><img src="https://img.shields.io/pypi/dm/scrapling" alt="PyPI package downloads"></a>
|
||||
<a href="https://github.com/D4Vinci/Scrapling/tree/main/agent-skill" alt="AI Agent Skill directory">
|
||||
<img alt="Static Badge" src="https://img.shields.io/badge/Skill-black?style=flat&label=Agent&link=https%3A%2F%2Fgithub.com%2FD4Vinci%2FScrapling%2Ftree%2Fmain%2Fagent-skill"></a>
|
||||
@@ -94,11 +96,11 @@ MySpider().start()
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/coldproxy.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://coldproxy.com/" target="_blank"><b>ColdProxy</b></a> fournit des proxies résidentiels et de datacenter pour un web scraping stable, la collecte de données publiques et des tests géolocalisés dans plus de 195 pays.
|
||||
<td> <a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank"><b>ColdProxy</b></a> fournit des proxies résidentiels et de datacenter pour un web scraping stable, la collecte de données publiques et des tests géolocalisés dans plus de 195 pays.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
|
||||
+4
-2
@@ -16,6 +16,8 @@
|
||||
<img alt="Tests" src="https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml/badge.svg"></a>
|
||||
<a href="https://badge.fury.io/py/Scrapling" alt="PyPI version">
|
||||
<img alt="PyPI version" src="https://badge.fury.io/py/Scrapling.svg"></a>
|
||||
<a href="https://hub.docker.com/r/pyd4vinci/scrapling" target="_blank">
|
||||
<img alt="Docker Pulls" src="https://img.shields.io/docker/pulls/pyd4vinci/scrapling?labelColor=%20%23FDB062&logo=Docker&labelColor=%20%23528bff"></a>
|
||||
<a href="https://clickpy.clickhouse.com/dashboard/scrapling" rel="nofollow"><img src="https://img.shields.io/pypi/dm/scrapling" alt="PyPI package downloads"></a>
|
||||
<a href="https://github.com/D4Vinci/Scrapling/tree/main/agent-skill" alt="AI Agent Skill directory">
|
||||
<img alt="Static Badge" src="https://img.shields.io/badge/Skill-black?style=flat&label=Agent&link=https%3A%2F%2Fgithub.com%2FD4Vinci%2FScrapling%2Ftree%2Fmain%2Fagent-skill"></a>
|
||||
@@ -94,11 +96,11 @@ MySpider().start()
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/coldproxy.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://coldproxy.com/" target="_blank"><b>ColdProxy</b></a> は、安定したウェブスクレイピング、公開データ収集、195以上の国・地域でのジオターゲティングテストのために、レジデンシャルおよびデータセンタープロキシを提供します。
|
||||
<td> <a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank"><b>ColdProxy</b></a> は、安定したウェブスクレイピング、公開データ収集、195以上の国・地域でのジオターゲティングテストのために、レジデンシャルおよびデータセンタープロキシを提供します。
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
|
||||
+4
-2
@@ -16,6 +16,8 @@
|
||||
<img alt="Tests" src="https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml/badge.svg"></a>
|
||||
<a href="https://badge.fury.io/py/Scrapling" alt="PyPI version">
|
||||
<img alt="PyPI version" src="https://badge.fury.io/py/Scrapling.svg"></a>
|
||||
<a href="https://hub.docker.com/r/pyd4vinci/scrapling" target="_blank">
|
||||
<img alt="Docker Pulls" src="https://img.shields.io/docker/pulls/pyd4vinci/scrapling?labelColor=%20%23FDB062&logo=Docker&labelColor=%20%23528bff"></a>
|
||||
<a href="https://clickpy.clickhouse.com/dashboard/scrapling" rel="nofollow"><img src="https://img.shields.io/pypi/dm/scrapling" alt="PyPI package downloads"></a>
|
||||
<a href="https://github.com/D4Vinci/Scrapling/tree/main/agent-skill" alt="AI Agent Skill directory">
|
||||
<img alt="Static Badge" src="https://img.shields.io/badge/Skill-black?style=flat&label=Agent&link=https%3A%2F%2Fgithub.com%2FD4Vinci%2FScrapling%2Ftree%2Fmain%2Fagent-skill"></a>
|
||||
@@ -94,11 +96,11 @@ MySpider().start()
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/coldproxy.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://coldproxy.com/" target="_blank"><b>ColdProxy</b></a>는 안정적인 웹 스크래핑, 공개 데이터 수집, 195개 이상의 국가에서의 지역 타겟팅 테스트를 위한 주거용 및 데이터센터 프록시를 제공합니다.
|
||||
<td> <a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank"><b>ColdProxy</b></a>는 안정적인 웹 스크래핑, 공개 데이터 수집, 195개 이상의 국가에서의 지역 타겟팅 테스트를 위한 주거용 및 데이터센터 프록시를 제공합니다.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
|
||||
@@ -18,6 +18,8 @@
|
||||
<img alt="Tests" src="https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml/badge.svg"></a>
|
||||
<a href="https://badge.fury.io/py/Scrapling" alt="PyPI version">
|
||||
<img alt="PyPI version" src="https://badge.fury.io/py/Scrapling.svg"></a>
|
||||
<a href="https://hub.docker.com/r/pyd4vinci/scrapling" target="_blank">
|
||||
<img alt="Docker Pulls" src="https://img.shields.io/docker/pulls/pyd4vinci/scrapling?labelColor=%20%23FDB062&logo=Docker&labelColor=%20%23528bff"></a>
|
||||
<a href="https://clickpy.clickhouse.com/dashboard/scrapling" rel="nofollow"><img src="https://img.shields.io/pypi/dm/scrapling" alt="PyPI package downloads"></a>
|
||||
<a href="https://github.com/D4Vinci/Scrapling/tree/main/agent-skill" alt="AI Agent Skill directory">
|
||||
<img alt="Static Badge" src="https://img.shields.io/badge/Skill-black?style=flat&label=Agent&link=https%3A%2F%2Fgithub.com%2FD4Vinci%2FScrapling%2Ftree%2Fmain%2Fagent-skill"></a>
|
||||
@@ -96,11 +98,11 @@ MySpider().start()
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/coldproxy.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> A <a href="https://coldproxy.com/" target="_blank"><b>ColdProxy</b></a> oferece proxies residenciais e de datacenter para web scraping estável, coleta de dados públicos e testes com segmentação geográfica em mais de 195 países.
|
||||
<td> A <a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank"><b>ColdProxy</b></a> oferece proxies residenciais e de datacenter para web scraping estável, coleta de dados públicos e testes com segmentação geográfica em mais de 195 países.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
|
||||
+4
-2
@@ -16,6 +16,8 @@
|
||||
<img alt="Tests" src="https://github.com/D4Vinci/Scrapling/actions/workflows/tests.yml/badge.svg"></a>
|
||||
<a href="https://badge.fury.io/py/Scrapling" alt="PyPI version">
|
||||
<img alt="PyPI version" src="https://badge.fury.io/py/Scrapling.svg"></a>
|
||||
<a href="https://hub.docker.com/r/pyd4vinci/scrapling" target="_blank">
|
||||
<img alt="Docker Pulls" src="https://img.shields.io/docker/pulls/pyd4vinci/scrapling?labelColor=%20%23FDB062&logo=Docker&labelColor=%20%23528bff"></a>
|
||||
<a href="https://clickpy.clickhouse.com/dashboard/scrapling" rel="nofollow"><img src="https://img.shields.io/pypi/dm/scrapling" alt="PyPI package downloads"></a>
|
||||
<a href="https://github.com/D4Vinci/Scrapling/tree/main/agent-skill" alt="AI Agent Skill directory">
|
||||
<img alt="Static Badge" src="https://img.shields.io/badge/Skill-black?style=flat&label=Agent&link=https%3A%2F%2Fgithub.com%2FD4Vinci%2FScrapling%2Ftree%2Fmain%2Fagent-skill"></a>
|
||||
@@ -94,11 +96,11 @@ MySpider().start()
|
||||
</tr>
|
||||
<tr>
|
||||
<td width="200">
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/coldproxy.png">
|
||||
</a>
|
||||
</td>
|
||||
<td> <a href="https://coldproxy.com/" target="_blank"><b>ColdProxy</b></a> предоставляет резидентные и дата-центровые прокси для стабильного веб-скрейпинга, сбора публичных данных и гео-таргетированного тестирования в более чем 195 странах.
|
||||
<td> <a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank"><b>ColdProxy</b></a> предоставляет резидентные и дата-центровые прокси для стабильного веб-скрейпинга, сбора публичных данных и гео-таргетированного тестирования в более чем 195 странах.
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
|
||||
+1
-1
@@ -59,7 +59,7 @@ MySpider().start()
|
||||
<a href="https://proxidize.com/?utm_source=github&utm_medium=sponsorship&utm_campaign=scrapling&utm_content=d4vinci" target="_blank" title="Clean Proxies with No Nonsense.">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/proxidize.png" class="ad">
|
||||
</a>
|
||||
<a href="https://coldproxy.com/" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<a href="https://coldproxy.com/?utm_source=scrapling&utm_medium=github&utm_campaign=coldproxy&utm_content=platinum_sponsor" target="_blank" title="Residential, IPv6 & Datacenter Proxies for Web Scraping">
|
||||
<img src="https://raw.githubusercontent.com/D4Vinci/Scrapling/main/images/coldproxy.png" class="ad">
|
||||
</a>
|
||||
<a href="https://hypersolutions.co/?utm_source=github&utm_medium=readme&utm_campaign=scrapling" target="_blank" title="Bot Protection Bypass API for Akamai, DataDome, Incapsula & Kasada">
|
||||
|
||||
+1
-1
@@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta"
|
||||
[project]
|
||||
name = "scrapling"
|
||||
# Static version instead of a dynamic version so we can get better layer caching while building docker, check the docker file to understand
|
||||
version = "0.4.11"
|
||||
version = "0.4.12"
|
||||
description = "Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!"
|
||||
readme = {file = "README.md", content-type = "text/markdown"}
|
||||
license = {file = "LICENSE"}
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
__author__ = "Karim Shoair (karim.shoair@pm.me)"
|
||||
__version__ = "0.4.11"
|
||||
__version__ = "0.4.12"
|
||||
__copyright__ = "Copyright (c) 2024 Karim Shoair"
|
||||
|
||||
from typing import Any, TYPE_CHECKING
|
||||
|
||||
@@ -56,7 +56,7 @@ class StorageSystemMixin(ABC): # pragma: no cover
|
||||
the docs for more info.
|
||||
:return: A dictionary of the unique properties
|
||||
"""
|
||||
raise NotImplementedError("Storage system must implement `save` method")
|
||||
raise NotImplementedError("Storage system must implement `retrieve` method")
|
||||
|
||||
@staticmethod
|
||||
@lru_cache(128, typed=True)
|
||||
@@ -123,7 +123,6 @@ class SQLiteStorageSystem(StorageSystemMixin):
|
||||
""",
|
||||
(url, identifier, dumps(element_data)),
|
||||
)
|
||||
self.cursor.fetchall()
|
||||
self.connection.commit()
|
||||
|
||||
def retrieve(self, identifier: str) -> Optional[Dict[str, Any]]:
|
||||
|
||||
+4
-6
@@ -300,7 +300,9 @@ class Selector(SelectorsGeneration):
|
||||
|
||||
ignored_elements: set[Any] = set()
|
||||
if ignore_tags:
|
||||
ignored_elements.update(self._root.iter(*ignore_tags))
|
||||
for element in self._root.iter(*ignore_tags):
|
||||
if element not in ignored_elements:
|
||||
ignored_elements.update(element.iter())
|
||||
|
||||
_all_strings = []
|
||||
|
||||
@@ -315,11 +317,7 @@ class Selector(SelectorsGeneration):
|
||||
return False
|
||||
|
||||
owner = parent.getparent() if text_node.is_tail else parent
|
||||
while owner is not None:
|
||||
if owner in ignored_elements:
|
||||
return False
|
||||
owner = owner.getparent()
|
||||
return True
|
||||
return owner not in ignored_elements
|
||||
|
||||
for text_node in cast(list[_ElementUnicodeResult], _find_all_text_nodes(self._root)):
|
||||
text = str(text_node)
|
||||
|
||||
+2
-2
@@ -14,12 +14,12 @@
|
||||
"mimeType": "image/png"
|
||||
}
|
||||
],
|
||||
"version": "0.4.11",
|
||||
"version": "0.4.12",
|
||||
"packages": [
|
||||
{
|
||||
"registryType": "pypi",
|
||||
"identifier": "scrapling",
|
||||
"version": "0.4.11",
|
||||
"version": "0.4.12",
|
||||
"runtimeHint": "uvx",
|
||||
"packageArguments": [
|
||||
{
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[metadata]
|
||||
name = scrapling
|
||||
version = 0.4.11
|
||||
version = 0.4.12
|
||||
author = Karim Shoair
|
||||
author_email = karim.shoair@pm.me
|
||||
description = Scrapling is an undetectable, powerful, flexible, high-performance Python library that makes Web Scraping easy and effortless as it should be!
|
||||
|
||||
@@ -210,6 +210,23 @@ class TestAdvancedSelectors:
|
||||
|
||||
assert node.get_all_text("\n", strip=True) == "string1\nstring2\nstring3\nstring4\nstring5\nstring6\nstring7"
|
||||
|
||||
def test_get_all_text_keeps_text_after_comment(self):
|
||||
"""Tail text following a comment or PI node must still be collected"""
|
||||
html = "<div>before<!-- c -->after<p>x</p></div>"
|
||||
assert Selector(html, keep_comments=True).get_all_text() == "before\nafter\nx"
|
||||
|
||||
def test_get_all_text_deep_nesting(self):
|
||||
"""Deeply nested trees collect every visible text node and skip ignored subtrees"""
|
||||
depth = 500
|
||||
html = "<div>" + "".join(f"<span>t{i}" for i in range(depth)) + "leaf" + "</span>" * depth + "end</div>"
|
||||
text = Selector(html).get_all_text(strip=True)
|
||||
assert text.startswith("t0\nt1\n")
|
||||
assert "leaf" in text and text.endswith("end")
|
||||
assert text.count("\n") == depth
|
||||
|
||||
ignored = "<div>keep<script>" + "<b>x</b>" * depth + "</script>done</div>"
|
||||
assert Selector(ignored).get_all_text(strip=True) == "keep\ndone"
|
||||
|
||||
|
||||
class TestTextHandlerAdvanced:
|
||||
"""Test advanced TextHandler functionality"""
|
||||
|
||||
Reference in New Issue
Block a user